{"source":"CompareLLM","asOf":"2026-09-23","methodology":"/methodology","deprecatedFields":[{"field":"metrics.elo","replacement":"ratings","reason":"LM Arena's preference rating is a reference observation, not CompareLLM's rating. Read `ratings` for our category Elo."},{"field":"metrics.elo_coding","replacement":"ratings","reason":"Same as metrics.elo, scoped to Arena's coding split."},{"field":"metrics.tokens_per_sec","replacement":"speed.throughputClass","reason":"Exact throughput remains here for compatibility; the headline is the class. A new major version moves exact values under `details`."},{"field":"metrics.ttft_ms","replacement":"speed.latencyClass","reason":"Same as metrics.tokens_per_sec, for time to first token."}],"policy":"Announced fields keep their current meaning, unit and nullability until a new major API version removes them. Meaning changes never happen in place.","models":[{"id":"claude-fable-5-1","name":"Claude Fable 5.1","provider":"Anthropic","providerId":"anthropic","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1620,"label":"Coding Reported rating","shortLabel":"Coding rating","basis":{"text":"Reported rating · partial coverage (75% of recipe inputs)","caution":false},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Reported","elo":1620,"coveragePct":75},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Reported","elo":1593,"coveragePct":50},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":48,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4810,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":10,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":50,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":50,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/claude-fable-5-1#specifications"},"detailUrl":"/models/claude-fable-5-1"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Mixed ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":61.986666666666665,"rating":1620,"coveragePct":75,"inputs":[{"observationId":"claude-fable-5-1--deepswe-v1-1--20260915b","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":67.4,"unit":"%","normalized":67.4,"configuredWeight":0.4,"effectiveWeight":0.5333333333333333,"contribution":35.94666666666667,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://deepswe.datacurve.ai/","publisher":"DeepSWE leaderboard","tier":"third_party","configuration":"DeepSWE v1.1 public leaderboard, reported as a 67.4% average over five trials. Recorded because the model had only one coding benchmark and so could not be rated on the category at all.","effort":"reported_best","asOf":"2026-09-15","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"fable-terminal-bench-v4-2026-09-01","benchmarkId":"terminal-bench-v4","name":"Terminal-Bench","version":"4.0","family":"terminal-work","value":55.8,"unit":"%","normalized":55.8,"configuredWeight":0.35,"effectiveWeight":0.4666666666666666,"contribution":26.039999999999996,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Complement repository work with terminal-based tasks.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","publisher":"Anthropic","tier":"vendor","configuration":"Vendor launch disclosure; Terminal-Bench 4.0; reported model configuration.","effort":"reported_best","asOf":"2026-09-01","retrievedAt":"2026-09-11","note":"Direct vendor-reported launch result.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["frontiercode"],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[{"observationId":"fable-hle-tools-2026-09-01","benchmarkId":"hle-tools-launch-2026-09","name":"Humanity's Last Exam with tools","version":"launch-2026-09-03","family":"broad-academic","value":65,"unit":"%","normalized":65,"configuredWeight":0.2,"effectiveWeight":1,"contribution":65,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Tool-assisted academic problems broaden the evidence, with a smaller weight because tools affect outcomes.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","publisher":"Anthropic","tier":"vendor","configuration":"Vendor launch disclosure; Humanity’s Last Exam with tools; reported model configuration.","effort":"reported_best","asOf":"2026-09-01","retrievedAt":"2026-09-11","note":"Tool-assisted result; not comparable to no-tools HLE.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["graduate-science","advanced-mathematics","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"claude-fable-5-1","categoryId":"reasoning","score":70,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5.1 scores 53, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-fable-5-1","categoryId":"writing","score":70,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5.1 scores 53, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":59.300000000000004,"rating":1593,"coveragePct":50,"inputs":[{"observationId":"fable-osworld-2-partial-2026-09-01","benchmarkId":"osworld-2-offline-2026-08-08","name":"OSWorld 2.0 offline partial score","version":"2026.08.08","family":"desktop-use","value":77.9,"unit":"%","normalized":77.9,"configuredWeight":0.3,"effectiveWeight":0.6,"contribution":46.74,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Desktop interaction is a complementary core agent workload; offline/partial-credit scope is disclosed.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","publisher":"Anthropic","tier":"vendor","configuration":"Vendor launch disclosure; OSWorld 2.0 offline partial-credit score.","effort":"reported_best","asOf":"2026-09-01","retrievedAt":"2026-09-11","note":"Partial-credit score; strict score is a different measurement.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"fable-automationbench-2026-09-01","benchmarkId":"automationbench-launch-2026-09","name":"AutomationBench","version":"launch-2026-09-03","family":"workflow-automation","value":31.4,"unit":"%","normalized":31.4,"configuredWeight":0.2,"effectiveWeight":0.4,"contribution":12.56,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Include multistep workflow automation.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","publisher":"Anthropic","tier":"vendor","configuration":"Vendor launch disclosure; AutomationBench; reported model configuration.","effort":"reported_best","asOf":"2026-09-01","retrievedAt":"2026-09-11","note":"Direct vendor-reported launch result.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["professional-computer-tasks","web-research"],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-fable-5-1","categoryId":"chat","score":70,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5.1 scores 53, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"claude-fable-5-1","categoryId":"vision_understanding","score":70,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5.1 scores 53, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1672,"coveragePct":23.863636363636363,"sourceLabel":"Mixed","contributors":[{"categoryId":"coding","rating":1620,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1700,"weight":0.22727272727272727},{"categoryId":"writing","rating":1700,"weight":0.13636363636363635},{"categoryId":"agents","rating":1593,"weight":0.13636363636363635},{"categoryId":"chat","rating":1700,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1700,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1620,"basisLine":"Reported · Coding · 2 benchmark families · 75% benchmark coverage","coveragePct":75,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1672,"metrics":{"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","verificationStatus":"verified","observedAt":"2026-09-11"},"cached_input_price_per_m":{"value":0.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","verificationStatus":"verified","observedAt":"2026-09-11"},"input_price_per_m":{"value":10,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","verificationStatus":"verified","observedAt":"2026-09-11"},"output_price_per_m":{"value":50,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","verificationStatus":"verified","observedAt":"2026-09-11"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://platform.claude.com/docs/en/models/fable-5-1/overview","verificationStatus":"verified","observedAt":"2026-09-11"},"tokens_per_sec":{"value":48,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5.1","verificationStatus":"verified","observedAt":"2026-09-13"},"ttft_ms":{"value":4810,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5.1","verificationStatus":"verified","observedAt":"2026-09-13"}}},{"id":"gpt-6-astra","name":"GPT-6 Astra","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1660,"label":"Coding Reported rating","shortLabel":"Coding rating","basis":{"text":"Reported rating · full coverage (100% of recipe inputs)","caution":false},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Reported","elo":1660,"coveragePct":100},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Reported","elo":1892,"coveragePct":100},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Reported","elo":1662,"coveragePct":100},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Reported","elo":1919,"coveragePct":100}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":36,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3560,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":10,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":50,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":50,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/gpt-6-astra#specifications"},"detailUrl":"/models/gpt-6-astra"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Mixed ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":66.03,"rating":1660,"coveragePct":100,"inputs":[{"observationId":"astra-deepswe-v1-1-2026-09-03","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":74.1,"unit":"%","normalized":74.1,"configuredWeight":0.4,"effectiveWeight":0.4,"contribution":29.64,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. ","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-terminal-bench-v4-2026-09-03","benchmarkId":"terminal-bench-v4","name":"Terminal-Bench","version":"4.0","family":"terminal-work","value":57.9,"unit":"%","normalized":57.9,"configuredWeight":0.35,"effectiveWeight":0.35,"contribution":20.264999999999997,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Complement repository work with terminal-based tasks.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. ","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-frontiercode-v1-1-extended-2026-09-03","benchmarkId":"frontiercode-v1-1-extended","name":"FrontierCode Extended","version":"1.1","family":"frontiercode","value":64.5,"unit":"%","normalized":64.5,"configuredWeight":0.25,"effectiveWeight":0.25,"contribution":16.125,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. One contribution for related code suites; prefer Extended for breadth, independently of achieved score.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Codex-like developer instructions; see source footnote 8.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":[],"excluded":[{"observationId":"astra-frontiercode-v1-1-main-2026-09-03","benchmarkId":"frontiercode-v1-1-main","value":53.3,"sourceUrl":"https://openai.com/index/gpt-6-astra/","reason":"Alternative/conflicting result; one contribution per family. Vendor priority, declared alternative order, date, then ID selects the result—not its score."}],"sourceLabel":"Reported"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":89.19,"rating":1892,"coveragePct":100,"inputs":[{"observationId":"astra-gpqa-diamond-launch-2026-09-2026-09-03","benchmarkId":"gpqa-diamond-launch-2026-09","name":"GPQA Diamond","version":"launch-2026-09-03","family":"graduate-science","value":96,"unit":"%","normalized":96,"configuredWeight":0.35,"effectiveWeight":0.35000000000000003,"contribution":33.6,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Broad scientific reasoning is a central signal.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Dataset revision not specified in launch table.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-frontiermath-tier4-v2-2026-09-03","benchmarkId":"frontiermath-tier4-v2","name":"FrontierMath Tier 4","version":"2","family":"advanced-mathematics","value":97.6,"unit":"%","normalized":97.6,"configuredWeight":0.35,"effectiveWeight":0.35000000000000003,"contribution":34.160000000000004,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Give equal weight to difficult mathematics rather than letting one discipline dominate.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. ","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-hle-tools-launch-2026-09-2026-09-03","benchmarkId":"hle-tools-launch-2026-09","name":"Humanity's Last Exam with tools","version":"launch-2026-09-03","family":"broad-academic","value":57.2,"unit":"%","normalized":57.2,"configuredWeight":0.2,"effectiveWeight":0.20000000000000004,"contribution":11.440000000000003,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Tool-assisted academic problems broaden the evidence, with a smaller weight because tools affect outcomes.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Tools enabled; dataset revision not specified.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-arc-agi-3-launch-2026-09-2026-09-03","benchmarkId":"arc-agi-3-launch-2026-09","name":"ARC-AGI-3","version":"launch-2026-09-03","family":"interactive-abstraction","value":99.9,"unit":"%","normalized":99.9,"configuredWeight":0.1,"effectiveWeight":0.10000000000000002,"contribution":9.990000000000002,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Interactive abstraction complements academic tasks; small weight limits harness-specific influence.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Responses API harness modifications disclosed in source footnote 1; measures the model with its harness.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":[],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"writing","label":"Writing","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[{"observationId":"astra-story-writing-a310070d-high","benchmarkId":"story-writing-estimated-win-a310070d","name":"Creative Story-Writing: published estimated win chance","version":"a310070d5ce19499b48f6d239bd2548622c4a6d0","family":"constrained-story-writing","value":92,"unit":"%","normalized":92,"configuredWeight":0.7,"effectiveWeight":1,"contribution":92,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Direct creative-writing evaluator evidence; scoped to constrained stories and a frozen comparison cohort.","weightSourceUrl":"https://github.com/lechmazur/writing/blob/a310070d5ce19499b48f6d239bd2548622c4a6d0/README.md","normalizationNote":"Use the publisher's bounded estimated win-chance column, not its unbounded comparison score. Fixed probability endpoints 0–100%; applies only to the pinned 50-model cohort and evaluator mix. This is not a confidence level or a head-to-head prediction for this website.","sourceUrl":"https://github.com/lechmazur/writing/blob/a310070d5ce19499b48f6d239bd2548622c4a6d0/README.md","publisher":"Lech Mazur","tier":"third_party","configuration":"High effort; evaluator-v2/v3 comparison evidence with bridge validation; 50-model cohort pinned to source commit. Model judges compare stories written to matching constrained briefs.","effort":"high","asOf":"2026-09-05","retrievedAt":"2026-09-08","note":"Published estimated win chance is 92%; separate relative comparison score is 3.5 (interval 3.4–3.6). These units are not interchangeable. Not an absolute writing grade; not a website pairwise prediction.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-6-astra","categoryId":"writing","score":70,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Astra (max) scores 53, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-astra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":66.14999999999999,"rating":1662,"coveragePct":100,"inputs":[{"observationId":"astra-agents-last-exam-launch-2026-09-2026-09-03","benchmarkId":"agents-last-exam-launch-2026-09","name":"Agents' Last Exam","version":"launch-2026-09-03","family":"professional-computer-tasks","value":59.3,"unit":"%","normalized":59.3,"configuredWeight":0.3,"effectiveWeight":0.3,"contribution":17.79,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. End-to-end professional tasks are central to agent usefulness.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. ","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-osworld-2-offline-2026-08-08-2026-09-03","benchmarkId":"osworld-2-offline-2026-08-08","name":"OSWorld 2.0 offline partial score","version":"2026.08.08","family":"desktop-use","value":72.6,"unit":"%","normalized":72.6,"configuredWeight":0.3,"effectiveWeight":0.3,"contribution":21.779999999999998,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Desktop interaction is a complementary core agent workload; offline/partial-credit scope is disclosed.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Offline subset and partial credit, not the full online task set.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-automationbench-launch-2026-09-2026-09-03","benchmarkId":"automationbench-launch-2026-09","name":"AutomationBench","version":"launch-2026-09-03","family":"workflow-automation","value":41.4,"unit":"%","normalized":41.4,"configuredWeight":0.2,"effectiveWeight":0.2,"contribution":8.28,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Include multistep workflow automation.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. ","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-browsecomp-launch-2026-09-2026-09-03","benchmarkId":"browsecomp-launch-2026-09","name":"BrowseComp","version":"launch-2026-09-03","family":"web-research","value":91.5,"unit":"%","normalized":91.5,"configuredWeight":0.2,"effectiveWeight":0.2,"contribution":18.3,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Include browsing and information retrieval without dominating desktop evidence.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. ","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":[],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[{"observationId":"astra-tau3-banking-aa-2026-09-03-max","benchmarkId":"tau3-banking-aa-2026-09-03","name":"τ³-Banking pass@1 (AA launch chart)","version":"aa-2026-09-03","family":"support-task-completion","value":41,"unit":"%","normalized":41,"configuredWeight":0.25,"effectiveWeight":1,"contribution":41,"reason":"Authored relevance policy: support outcomes receive 25%. This is the sole present family for Astra, so its Chat rating is provisional and reflects banking-support task completion only, not general conversation quality.","weightSourceUrl":"https://artificialanalysis.ai/evaluations/tau3-banking","normalizationNote":"Fixed 0–100 task-success percentage. Published chart rounds to whole percentages; preserve 41 without inventing decimals. Measures banking task outcomes, not conversational style or general chat satisfaction.","sourceUrl":"https://artificialanalysis.ai/articles/benchmarking-gpt-6-astra","publisher":"Artificial Analysis","tier":"third_party","configuration":"GPT-6 Astra (max), τ³-Banking panel in the article's Intelligence Evaluations chart. AA describes 97 banking-support tasks, pass@1 averaged over repeats and backend-state grading. Exact historical repeat count/harness revision unspecified.","effort":"max","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Provisional chat/support proxy, 25% recipe coverage. Visually verified label: 41%. Source chart: https://cdn.sanity.io/images/6vfeftx9/articles/086540355fe9876fc4106c484b6066fd819e8205-2924x4184.png . Banking task success does not measure conversational tone, helpfulness or general satisfaction. Whole-percent source precision retained.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["conversation-quality","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-6-astra","categoryId":"chat","score":70,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Astra (max) scores 53, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-astra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":91.92,"rating":1919,"coveragePct":100,"inputs":[{"observationId":"astra-screenspot-pro-no-tools-launch-2026-09-03","benchmarkId":"screenspot-pro-no-tools-launch","name":"ScreenSpot-Pro no tools","version":"launch-2026-09-03","family":"visual-grounding","value":92.7,"unit":"%","normalized":92.7,"configuredWeight":0.5,"effectiveWeight":0.5,"contribution":46.35,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Direct no-tool localization gets the largest weight.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. No tools; interface localization is a limited part of vision understanding.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-benchcad-launch-2026-09-2026-09-03","benchmarkId":"benchcad-launch-2026-09","name":"BenchCAD","version":"launch-2026-09-03","family":"spatial-reconstruction","value":95.9,"unit":"%","normalized":95.9,"configuredWeight":0.3,"effectiveWeight":0.3,"contribution":28.77,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Add spatial interpretation with a smaller share because code/tools contribute.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Tool-assisted geometric overlap. Image interpretation plus code generation, not native image generation.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"astra-openscore-string-quartets-launch-2026-09-03","benchmarkId":"openscore-string-quartets-launch","name":"OpenScore String Quartets 1 - OMR-NED","version":"launch-2026-09-03","family":"visual-symbol-recognition","value":0.84,"unit":"fraction","normalized":84,"configuredWeight":0.2,"effectiveWeight":0.2,"contribution":16.8,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Symbol recognition broadens this recipe; the narrow music-notation domain gets a modest share.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published complement of normalized edit distance, using fixed 0–1 endpoints.","sourceUrl":"https://openai.com/index/gpt-6-astra/","publisher":"OpenAI","tier":"vendor","configuration":"Publisher-reported best across effort settings; research/API environment. Exact per-row effort and full run configuration were not disclosed. Optical score recognition, not audio understanding or music generation.","effort":"reported_best","asOf":"2026-09-03","retrievedAt":"2026-09-08","note":"Proposed accepted input in an unpublished draft. Scope/version is limited to the cited launch result; does not assert an exact undisclosed dataset revision.","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":[],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1783,"coveragePct":68.18181818181817,"sourceLabel":"Mixed","contributors":[{"categoryId":"coding","rating":1660,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1892,"weight":0.22727272727272727},{"categoryId":"writing","rating":1700,"weight":0.13636363636363635},{"categoryId":"agents","rating":1662,"weight":0.13636363636363635},{"categoryId":"chat","rating":1700,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1919,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1660,"basisLine":"Reported · Coding · 3 benchmark families · 100% benchmark coverage","coveragePct":100,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1783,"metrics":{"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://developers.openai.com/api/docs/models/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-11"},"cached_input_price_per_m":{"value":1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://developers.openai.com/api/docs/models/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-11"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://developers.openai.com/api/docs/models/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-11"},"input_price_per_m":{"value":10,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://developers.openai.com/api/docs/models/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-11"},"output_price_per_m":{"value":50,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://developers.openai.com/api/docs/models/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-11"},"tokens_per_sec":{"value":36,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-13"},"ttft_ms":{"value":3560,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-astra","verificationStatus":"verified","observedAt":"2026-09-13"}}},{"id":"gpt-image-2-5-sunburst","name":"GPT Image 2.5 Sunburst","provider":"OpenAI","providerId":"openai","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["text"],"outputs":["image"],"hasVision":false,"openWeights":false},"headline":{"value":1870,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1870,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":33.8,"unit":"s","kind":"speed"}],"speedClass":"30s+","price":null,"specifications":{"count":11,"published":10,"detailUrl":"/models/gpt-image-2-5-sunburst#specifications"},"detailUrl":"/models/gpt-image-2-5-sunburst"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":87,"rating":1870,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"gpt-image-2-5-sunburst","categoryId":"image_generation","score":87,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where GPT Image 2.5 Sunburst (max) scores 1182.31. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1870,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1870,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"image_generation":{"elo":1870,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1870,"metrics":{"image_gen_time_sec":{"value":33.8,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://www.datastudios.org/post/openai-chatgpt-images-2-5-flare-sunburst-api-pricing","verificationStatus":"verified","observedAt":"2026-09-14"}}},{"id":"gpt-image-2-5-flare","name":"GPT Image 2.5 Flare","provider":"OpenAI","providerId":"openai","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["text"],"outputs":["image"],"hasVision":false,"openWeights":false},"headline":{"value":1880,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1880,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":27.8,"unit":"s","kind":"speed"}],"speedClass":"15–30s","price":null,"specifications":{"count":11,"published":10,"detailUrl":"/models/gpt-image-2-5-flare#specifications"},"detailUrl":"/models/gpt-image-2-5-flare"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":88,"rating":1880,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"gpt-image-2-5-flare","categoryId":"image_generation","score":88,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where GPT Image 2.5 Flare (max) scores 1187.74. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1880,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1880,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"image_generation":{"elo":1880,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1880,"metrics":{"image_gen_time_sec":{"value":27.8,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://www.datastudios.org/post/openai-chatgpt-images-2-5-flare-sunburst-api-pricing","verificationStatus":"verified","observedAt":"2026-09-14"}}},{"id":"seedance-2-5","name":"Seedance 2.5","provider":"ByteDance","providerId":"bytedance","presentation":{"task":"video_generation","taskLabel":"Video generation","modality":"video","capabilities":{"inputs":["text"],"outputs":["video"],"hasVision":false,"openWeights":false},"headline":{"value":1800,"label":"Video Estimated rating","shortLabel":"Video rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"video_generation","isOverall":false},"categories":[{"categoryId":"video_generation","label":"Video generation","state":"rated","sourceLabel":"Estimated","elo":1800,"coveragePct":0}],"facts":[],"speedClass":null,"price":null,"specifications":{"count":9,"published":8,"detailUrl":"/models/seedance-2-5#specifications"},"detailUrl":"/models/seedance-2-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"rated","score":80,"rating":1800,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"seedance-2-5","categoryId":"video_generation","score":80,"rationale":"Proposed editorial score: hands-on examples support strong continuity and physical interactions, while long-shot motion timing and audio remain imperfect. The reviewer sells video tooling, and the small sample is not a controlled benchmark.","sourceUrls":["https://seed.bytedance.com/en/blog/one-take-creation-flexible-referencing-introducing-seedance-2-5","https://www.kapwing.com/resources/is-seedance-2-5-actually-better-than-2-0-heres-what-i-found/"],"confidence":"low","asOf":"2026-09-12","reviewOn":"2026-09-26"},"sourceLabel":"Estimated"},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1800,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"video_generation","rating":1800,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"video_generation":{"elo":1800,"basisLine":"Estimated · Video generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-09-26","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1800,"metrics":{}},{"id":"minimax-h3","name":"MiniMax H3","provider":"MiniMax","providerId":"minimax","presentation":{"task":"video_generation","taskLabel":"Video generation","modality":"video","capabilities":{"inputs":["text"],"outputs":["video"],"hasVision":false,"openWeights":false},"headline":{"value":1870,"label":"Video Estimated rating","shortLabel":"Video rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"video_generation","isOverall":false},"categories":[{"categoryId":"video_generation","label":"Video generation","state":"rated","sourceLabel":"Estimated","elo":1870,"coveragePct":0}],"facts":[{"metric":"video_gen_time_sec","label":"Gen Time","value":3,"unit":"s","kind":"speed"},{"metric":"video_price_per_sec","label":"Video $","value":0.08,"unit":"$/sec","kind":"cost"}],"speedClass":"<30s","price":{"value":0.08,"unit":"USD/output second"},"specifications":{"count":8,"published":8,"detailUrl":"/models/minimax-h3#specifications"},"detailUrl":"/models/minimax-h3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"rated","score":87,"rating":1870,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"minimax-h3","categoryId":"video_generation","score":87,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where MiniMax H3 scores 1220. Placed by where that sits between Seedance 1.5 pro at 1000 and Gemini Omni Flash at 1233, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/video/leaderboard/text-to-video"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1870,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"video_generation","rating":1870,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"video_generation":{"elo":1870,"basisLine":"Estimated · Video generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1870,"metrics":{"video_price_per_sec":{"value":0.08,"unit":"$/sec","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://platform.minimax.io/docs/guides/pricing-paygo","verificationStatus":"verified","observedAt":"2026-09-13"},"video_gen_time_sec":{"value":3,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://www.minimax.io/blog/minimax-h3","verificationStatus":"verified","observedAt":"2026-09-14"}}},{"id":"suno-v6","name":"Suno v6","provider":"Suno","providerId":"suno","presentation":{"task":"music_generation","taskLabel":"Music generation","modality":"audio","capabilities":{"inputs":["text"],"outputs":["audio"],"hasVision":false,"openWeights":false},"headline":{"value":1700,"label":"Music Estimated rating","shortLabel":"Music rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"music_generation","isOverall":false},"categories":[{"categoryId":"music_generation","label":"Music generation","state":"rated","sourceLabel":"Estimated","elo":1700,"coveragePct":0}],"facts":[],"speedClass":null,"price":null,"specifications":{"count":5,"published":4,"detailUrl":"/models/suno-v6#specifications"},"detailUrl":"/models/suno-v6"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"rated","score":70,"rating":1700,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"suno-v6","categoryId":"music_generation","score":70,"rationale":"Proposed editorial score: early feedback mixes cleaner production and improved control with concerns over emotional vocals, phrasing and preservation during edits. Conservative launch-week opinion, not a popularity score or listening benchmark.","sourceUrls":["https://suno.com/release-notes","https://www.reddit.com/r/SunoAI/comments/1wcrqea/suno_v6_an_honest_review_from_a_producer_whos/","https://www.reddit.com/r/SunoAI/comments/1wbs7t2/v6_sounds_better_but_the_music_feels_worse_as_a/"],"confidence":"low","asOf":"2026-09-12","reviewOn":"2026-09-26"},"sourceLabel":"Estimated"}],"overall":null},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"music_generation":{"elo":1700,"basisLine":"Estimated · Music generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-09-26","coveragePct":0,"opinion":true}},"speed":{},"overallElo":null,"metrics":{}},{"id":"granite-speech-5-0-470m-turboctc","name":"Granite Speech 5.0 470M TurboCTC","provider":"IBM","providerId":"ibm","presentation":{"task":"speech_recognition","taskLabel":"Speech recognition","modality":"audio","capabilities":{"inputs":["audio"],"outputs":["text"],"hasVision":false,"openWeights":true},"headline":{"value":1750,"label":"ASR Estimated rating","shortLabel":"ASR rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"speech_recognition","isOverall":false},"categories":[{"categoryId":"speech_recognition","label":"Speech recognition","state":"rated","sourceLabel":"Estimated","elo":1750,"coveragePct":0}],"facts":[{"metric":"asr_rtfx","label":"RTFx","value":13042.98,"unit":"RTFx","kind":"speed"}],"speedClass":"1×+","price":null,"specifications":{"count":6,"published":4,"detailUrl":"/models/granite-speech-5-0-470m-turboctc#specifications"},"detailUrl":"/models/granite-speech-5-0-470m-turboctc"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"rated","score":75,"rating":1750,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"granite-speech-5-0-470m-turboctc","categoryId":"speech_recognition","score":75,"rationale":"Proposed editorial score for English speech recognition only: vendor evaluation claims and early practitioner interest indicate useful transcription quality, but benchmark numbers have not been imported or independently reproduced here. No multilingual competence is implied.","sourceUrls":["https://huggingface.co/ibm-granite/granite-speech-5.0-470m-turboctc","https://www.reddit.com/r/speechtech/comments/1w2i9t6/extremely_fast_and_accurate_transcription_with/"],"confidence":"low","asOf":"2026-09-12","reviewOn":"2026-09-26"},"sourceLabel":"Estimated"},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":null},"isFrontier":true,"isOpenSource":true,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"speech_recognition":{"elo":1750,"basisLine":"Estimated · Speech recognition · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-09-26","coveragePct":0,"opinion":true}},"speed":{},"overallElo":null,"metrics":{"asr_rtfx":{"value":13042.98,"unit":"RTFx","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://huggingface.co/ibm-granite/granite-speech-5.0-470m-turboctc","verificationStatus":"verified","observedAt":"2026-08-25"}}},{"id":"gemini-3-8-flash","name":"Gemini 3.8 Flash","provider":"Google","providerId":"google","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1482,"label":"Coding Reported rating","shortLabel":"Coding rating","basis":{"text":"Reported rating · partial coverage (75% of recipe inputs)","caution":false},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Reported","elo":1482,"coveragePct":75},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":114,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2210,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.75,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":3.75,"unit":"$/1M tok","kind":"cost"}],"speedClass":"100–200 tok/s","price":{"value":3.75,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/gemini-3-8-flash#specifications"},"detailUrl":"/models/gemini-3-8-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Mixed ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":48.22,"rating":1482,"coveragePct":75,"inputs":[{"observationId":"gemini-3-8-flash--deepswe-v1-1--20260915","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":73.7,"unit":"%","normalized":73.7,"configuredWeight":0.4,"effectiveWeight":0.5333333333333333,"contribution":39.306666666666665,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://www.vellum.ai/blog/gemini-3-8-flash-benchmarks-explained","publisher":"Google","tier":"vendor","configuration":"Google's own launch table for Gemini 3.8 Flash, read through secondary reporting of that table. DeepSWE v1.1 end-to-end agentic solve rate.","effort":"reported_best","asOf":"2026-09-02","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"gemini-3-8-flash--terminal-bench-v4--20260915","benchmarkId":"terminal-bench-v4","name":"Terminal-Bench","version":"4.0","family":"terminal-work","value":19.1,"unit":"%","normalized":19.1,"configuredWeight":0.35,"effectiveWeight":0.4666666666666666,"contribution":8.913333333333332,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Complement repository work with terminal-based tasks.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://www.vellum.ai/blog/gemini-3-8-flash-benchmarks-explained","publisher":"Google","tier":"vendor","configuration":"Terminal-Bench 4.0, from the same Google launch table. The model's widely quoted 89.4 is Terminal-Bench 2.1, a different and much easier scale; the two are not interchangeable.","effort":"reported_best","asOf":"2026-09-02","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["frontiercode"],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gemini-3-8-flash","categoryId":"reasoning","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.8 Flash (high) scores 41, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-8-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-8-flash","categoryId":"writing","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.8 Flash (high) scores 41, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-8-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[{"observationId":"gemini-3-8-flash--osworld-2-offline-2026-08-08--20260915","benchmarkId":"osworld-2-offline-2026-08-08","name":"OSWorld 2.0 offline partial score","version":"2026.08.08","family":"desktop-use","value":59,"unit":"%","normalized":59,"configuredWeight":0.3,"effectiveWeight":1,"contribution":59,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Desktop interaction is a complementary core agent workload; offline/partial-credit scope is disclosed.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://www.vellum.ai/blog/gemini-3-8-flash-benchmarks-explained","publisher":"Google","tier":"vendor","configuration":"OSWorld 2.0 offline partial score from Google's launch table.","effort":"reported_best","asOf":"2026-09-02","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["professional-computer-tasks","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gemini-3-8-flash","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.8 Flash (high) scores 41, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-8-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-8-flash","categoryId":"chat","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.8 Flash (high) scores 41, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-8-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gemini-3-8-flash","categoryId":"vision_understanding","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.8 Flash (high) scores 41, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-8-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1630,"coveragePct":17.045454545454543,"sourceLabel":"Mixed","contributors":[{"categoryId":"coding","rating":1482,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1660,"weight":0.22727272727272727},{"categoryId":"writing","rating":1660,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1660,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1660,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1482,"basisLine":"Reported · Coding · 2 benchmark families · 75% benchmark coverage","coveragePct":75,"opinion":true}},"speed":{"throughputClass":"100–200 tok/s","latencyClass":"2s+"},"overallElo":1630,"metrics":{"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"input_price_per_m":{"value":0.75,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"cached_input_price_per_m":{"value":0.075,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"output_price_per_m":{"value":3.75,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"max_output_tokens":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"tokens_per_sec":{"value":114,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"ttft_ms":{"value":2210,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.8-flash","verificationStatus":"verified","observedAt":"2026-09-15"}}},{"id":"grok-4-6","name":"Grok 4.6","provider":"xAI","providerId":"xai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1641,"label":"Coding Reported rating","shortLabel":"Coding rating","basis":{"text":"Reported rating · thin coverage (65% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Reported","elo":1641,"coveragePct":65},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":500000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":48,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1750,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":6,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":6,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/grok-4-6#specifications"},"detailUrl":"/models/grok-4-6"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Mixed ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":64.13076923076923,"rating":1641,"coveragePct":65,"inputs":[{"observationId":"grok-4-6--deepswe-v1-1--20260915","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":65.9,"unit":"%","normalized":65.9,"configuredWeight":0.4,"effectiveWeight":0.6153846153846154,"contribution":40.55384615384616,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://x.ai/news/grok-4-6","publisher":"xAI","tier":"vendor","configuration":"xAI's Grok 4.6 announcement. DeepSWE v1.1 solve rate.","effort":"reported_best","asOf":"2026-08-12","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"grok-4-6--frontiercode-v1-1-extended--20260915","benchmarkId":"frontiercode-v1-1-extended","name":"FrontierCode Extended","version":"1.1","family":"frontiercode","value":61.3,"unit":"%","normalized":61.3,"configuredWeight":0.25,"effectiveWeight":0.3846153846153846,"contribution":23.576923076923073,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. One contribution for related code suites; prefer Extended for breadth, independently of achieved score.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://x.ai/news/grok-4-6","publisher":"xAI","tier":"vendor","configuration":"FrontierCode v1.1 Extended split, from xAI's announcement table.","effort":"reported_best","asOf":"2026-08-12","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["terminal-work"],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"grok-4-6","categoryId":"reasoning","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.6 (high) scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-6"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-6","categoryId":"writing","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.6 (high) scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-6"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"grok-4-6","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.6 (high) scores 44, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-6"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-6","categoryId":"chat","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.6 (high) scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-6"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"grok-4-6","categoryId":"vision_understanding","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.6 (high) scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-6"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1660,"coveragePct":14.772727272727272,"sourceLabel":"Mixed","contributors":[{"categoryId":"coding","rating":1641,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1670,"weight":0.22727272727272727},{"categoryId":"writing","rating":1670,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1670,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1670,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1641,"basisLine":"Reported · Coding · 2 benchmark families · 65% benchmark coverage","coveragePct":65,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1660,"metrics":{"cached_input_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"},"context_window":{"value":500000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"},"max_output_tokens":{"value":450000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"},"output_price_per_m":{"value":6,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"},"tokens_per_sec":{"value":48,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"},"ttft_ms":{"value":1750,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.6","verificationStatus":"verified","observedAt":"2026-09-15"}}},{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","provider":"DeepSeek","providerId":"deepseek","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":true},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":142,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1000,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.15,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.6,"unit":"$/1M tok","kind":"cost"}],"speedClass":"100–200 tok/s","price":{"value":0.6,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/deepseek-v4-1-flash#specifications"},"detailUrl":"/models/deepseek-v4-1-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[{"observationId":"deepseek-v4-1-flash--deepswe-v1-1--20260915","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":74.2,"unit":"%","normalized":74.2,"configuredWeight":0.4,"effectiveWeight":1,"contribution":74.2,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash","publisher":"DeepSeek","tier":"vendor","configuration":"DeepSeek's V4.1 Flash model card and technical report. DeepSWE v1.1 solve rate.","effort":"reported_best","asOf":"2026-09-10","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"deepseek-v4-1-flash","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4.1 Flash (max) scores 40, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-1-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[{"observationId":"deepseek-v4-1-flash--gpqa-diamond-launch-2026-09--20260915","benchmarkId":"gpqa-diamond-launch-2026-09","name":"GPQA Diamond","version":"launch-2026-09-03","family":"graduate-science","value":90.9,"unit":"%","normalized":90.9,"configuredWeight":0.35,"effectiveWeight":1,"contribution":90.9,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Broad scientific reasoning is a central signal.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash","publisher":"DeepSeek","tier":"vendor","configuration":"GPQA Diamond accuracy from the same model card.","effort":"reported_best","asOf":"2026-09-10","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"deepseek-v4-1-flash","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4.1 Flash (max) scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-1-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"deepseek-v4-1-flash","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4.1 Flash (max) scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-1-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"deepseek-v4-1-flash","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4.1 Flash (max) scores 40, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-1-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"deepseek-v4-1-flash","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4.1 Flash (max) scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-1-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"deepseek-v4-1-flash","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4.1 Flash (max) scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-1-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1635,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":true,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 1 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"100–200 tok/s","latencyClass":"500ms–2s"},"overallElo":1635,"metrics":{"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"output_price_per_m":{"value":0.6,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"max_output_tokens":{"value":384000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"cached_input_price_per_m":{"value":0.003,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"tokens_per_sec":{"value":142,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"},"ttft_ms":{"value":1000,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4.1-flash","verificationStatus":"verified","observedAt":"2026-09-15"}}},{"id":"glm-5-3","name":"GLM-5.3","provider":"Z.ai","providerId":"zai","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":true},"headline":{"value":1552,"label":"Coding Reported rating","shortLabel":"Coding rating","basis":{"text":"Reported rating · partial coverage (75% of recipe inputs)","caution":false},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Reported","elo":1552,"coveragePct":75},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":33,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4250,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.4,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.4,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":4.4,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/glm-5-3#specifications"},"detailUrl":"/models/glm-5-3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Mixed ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":55.18666666666667,"rating":1552,"coveragePct":75,"inputs":[{"observationId":"glm-5-3--deepswe-v1-1--20260915","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":66.9,"unit":"%","normalized":66.9,"configuredWeight":0.4,"effectiveWeight":0.5333333333333333,"contribution":35.68,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://huggingface.co/zai-org/GLM-5.3","publisher":"Z.ai","tier":"vendor","configuration":"Z.ai's own launch table, which reports DeepSWE v1.1 rising from 46.2 on GLM-5.2 to 66.9. Vendor-reported and not independently re-run.","effort":"reported_best","asOf":"2026-08-18","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."},{"observationId":"glm-5-3--terminal-bench-v4--20260915","benchmarkId":"terminal-bench-v4","name":"Terminal-Bench","version":"4.0","family":"terminal-work","value":41.8,"unit":"%","normalized":41.8,"configuredWeight":0.35,"effectiveWeight":0.4666666666666666,"contribution":19.506666666666664,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Complement repository work with terminal-based tasks.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://codingfleet.com/blog/terminal-bench-leaderboard-2026/","publisher":"Terminal-Bench","tier":"third_party","configuration":"Official tbench.ai Terminal-Bench 4.0 run, not the vendor's own 3.0 figure of 28.3.","effort":"reported_best","asOf":"2026-09-02","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["frontiercode"],"excluded":[],"sourceLabel":"Reported"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"glm-5-3","categoryId":"reasoning","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM-5.3 (max) scores 45, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"glm-5-3","categoryId":"writing","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM-5.3 (max) scores 45, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"glm-5-3","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM-5.3 (max) scores 45, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"glm-5-3","categoryId":"chat","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM-5.3 (max) scores 45, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1642,"coveragePct":18.75,"sourceLabel":"Mixed","contributors":[{"categoryId":"coding","rating":1552,"weight":0.25},{"categoryId":"reasoning","rating":1670,"weight":0.25},{"categoryId":"writing","rating":1670,"weight":0.15},{"categoryId":"agents","rating":1640,"weight":0.15},{"categoryId":"chat","rating":1670,"weight":0.2}]}},"isFrontier":true,"isOpenSource":true,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1552,"basisLine":"Reported · Coding · 2 benchmark families · 75% benchmark coverage","coveragePct":75,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1642,"metrics":{"cached_input_price_per_m":{"value":0.26,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"},"input_price_per_m":{"value":1.4,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"},"output_price_per_m":{"value":4.4,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"},"tokens_per_sec":{"value":33,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"},"ttft_ms":{"value":4250,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3","verificationStatus":"verified","observedAt":"2026-09-15"}}},{"id":"qwen3-8-max-0902","name":"Qwen3.8 Max (0902)","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":37,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2240,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":6,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":6,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/qwen3-8-max-0902#specifications"},"detailUrl":"/models/qwen3-8-max-0902"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[{"observationId":"qwen3-8-max-0902--deepswe-v1-1--20260915","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":57,"unit":"%","normalized":56.99999999999999,"configuredWeight":0.4,"effectiveWeight":1,"contribution":56.99999999999999,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://benchlm.ai/benchmarks/deepswe","publisher":"DeepSWE leaderboard","tier":"third_party","configuration":"Third-party DeepSWE v1.1 leaderboard, reported as 57% with a stated 3-point interval. The leaderboard lists the Qwen3.8 Max family without a snapshot date; this catalog entry is the 0902 snapshot.","effort":"reported_best","asOf":"2026-09-15","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-8-max-0902","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 Max scores 40, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-8-max-0902","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 Max scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-8-max-0902","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 Max scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-8-max-0902","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 Max scores 40, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-8-max-0902","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 Max scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"qwen3-8-max-0902","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 Max scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1635,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 1 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1635,"metrics":{"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"},"output_price_per_m":{"value":6,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"},"cached_input_price_per_m":{"value":0.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"},"tokens_per_sec":{"value":37,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"},"ttft_ms":{"value":2240,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-max-0902","verificationStatus":"verified","observedAt":"2026-09-15"}}},{"id":"muse-spark-1-3","name":"Muse Spark 1.3","provider":"Meta","providerId":"meta","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1620,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":84,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3500,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.25,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.25,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":4.25,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/muse-spark-1-3#specifications"},"detailUrl":"/models/muse-spark-1-3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[{"observationId":"muse-spark-1-3--deepswe-v1-1--20260915","benchmarkId":"deepswe-v1-1","name":"DeepSWE","version":"1.1","family":"repository-engineering","value":75.4,"unit":"%","normalized":75.4,"configuredWeight":0.4,"effectiveWeight":1,"contribution":75.4,"reason":"Admin-authored relevance policy, not empirical difficulty calibration. Repository-level work receives the largest share for coding relevance.","weightSourceUrl":"https://openai.com/index/gpt-6-astra/","normalizationNote":"Published percentage scale with fixed 0–100 endpoints; no cohort min/max calibration.","sourceUrl":"https://www.datacamp.com/blog/muse-spark-1-3","publisher":"Meta","tier":"vendor","configuration":"Meta's published Muse Spark 1.3 benchmarks, read through secondary reporting of that table. DeepSWE v1.1 solve rate.","effort":"reported_best","asOf":"2026-09-02","retrievedAt":"2026-09-15","note":"","selectionReason":"Applicable accepted result; own test, then vendor, then third party; declared alternative order, latest report date, then stable ID. Score does not determine selection."}],"missingFamilies":["terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"muse-spark-1-3","categoryId":"coding","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.3 (max) scores 48, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"muse-spark-1-3","categoryId":"reasoning","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.3 (max) scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-spark-1-3","categoryId":"writing","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.3 (max) scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"muse-spark-1-3","categoryId":"agents","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.3 (max) scores 48, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-spark-1-3","categoryId":"chat","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.3 (max) scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"muse-spark-1-3","categoryId":"vision_understanding","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.3 (max) scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1664,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1620,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1680,"weight":0.22727272727272727},{"categoryId":"writing","rating":1680,"weight":0.13636363636363635},{"categoryId":"agents","rating":1650,"weight":0.13636363636363635},{"categoryId":"chat","rating":1680,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1680,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1620,"basisLine":"Estimated · Coding · 1 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1664,"metrics":{"max_output_tokens":{"value":943718,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"},"input_price_per_m":{"value":1.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"},"cached_input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"},"output_price_per_m":{"value":4.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"},"tokens_per_sec":{"value":84,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"},"ttft_ms":{"value":3500,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.3","verificationStatus":"verified","observedAt":"2026-09-15"}}},{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","provider":"Google","providerId":"google","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":168,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1820,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.75,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":3.75,"unit":"$/1M tok","kind":"cost"}],"speedClass":"100–200 tok/s","price":{"value":3.75,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/gemini-3-7-flash#specifications"},"detailUrl":"/models/gemini-3-7-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gemini-3-7-flash","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.7 Flash (high) scores 39, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gemini-3-7-flash","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.7 Flash (high) scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-7-flash","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.7 Flash (high) scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gemini-3-7-flash","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.7 Flash (high) scores 39, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-7-flash","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.7 Flash (high) scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gemini-3-7-flash","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.7 Flash (high) scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1635,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"100–200 tok/s","latencyClass":"500ms–2s"},"overallElo":1635,"metrics":{"input_price_per_m":{"value":0.75,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":3.75,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.075,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":168,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1820,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"granite-4-2-8b","name":"Granite 4.2 8B","provider":"IBM","providerId":"ibm","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":true},"headline":{"value":1480,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1480,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":131072,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":62,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":220,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.1,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.15,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":0.15,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/granite-4-2-8b#specifications"},"detailUrl":"/models/granite-4-2-8b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":48,"rating":1480,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"granite-4-2-8b","categoryId":"coding","score":48,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Granite 4.2 8B scores 12, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/granite-4-2-8b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"granite-4-2-8b","categoryId":"reasoning","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Granite 4.2 8B scores 12, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/granite-4-2-8b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"granite-4-2-8b","categoryId":"writing","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Granite 4.2 8B scores 12, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/granite-4-2-8b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"granite-4-2-8b","categoryId":"agents","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Granite 4.2 8B scores 12, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/granite-4-2-8b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"granite-4-2-8b","categoryId":"chat","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Granite 4.2 8B scores 12, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/granite-4-2-8b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1542,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1480,"weight":0.25},{"categoryId":"reasoning","rating":1550,"weight":0.25},{"categoryId":"writing","rating":1550,"weight":0.15},{"categoryId":"agents","rating":1590,"weight":0.15},{"categoryId":"chat","rating":1550,"weight":0.2}]}},"isFrontier":true,"isOpenSource":true,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1480,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"200–500ms"},"overallElo":1542,"metrics":{"max_output_tokens":{"value":117964,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.05,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":62,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":220,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/ibm-granite/granite-4.2-8b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","provider":"Meta","providerId":"meta","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":true},"headline":{"value":1570,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1570,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":131072,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":83,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":250,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.1,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":1.1,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":10,"detailUrl":"/models/muse-glimmer-30b#specifications"},"detailUrl":"/models/muse-glimmer-30b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":57,"rating":1570,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"muse-glimmer-30b","categoryId":"coding","score":57,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Glimmer 30B (high) scores 35, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/articles/muse-glimmer"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"muse-glimmer-30b","categoryId":"reasoning","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Glimmer 30B (high) scores 35, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/articles/muse-glimmer"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-glimmer-30b","categoryId":"writing","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Glimmer 30B (high) scores 35, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/articles/muse-glimmer"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"muse-glimmer-30b","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Glimmer 30B (high) scores 35, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/articles/muse-glimmer"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-glimmer-30b","categoryId":"chat","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Glimmer 30B (high) scores 35, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/articles/muse-glimmer"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"muse-glimmer-30b","categoryId":"vision_understanding","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Glimmer 30B (high) scores 35, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/articles/muse-glimmer"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1625,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1570,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1640,"weight":0.22727272727272727},{"categoryId":"writing","rating":1640,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1640,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1640,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":true,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1570,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"200–500ms"},"overallElo":1625,"metrics":{"output_price_per_m":{"value":1.1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.04,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":117964,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":83,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":250,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-glimmer-30b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"ling-3-0-flash-vl","name":"Ling 3.0 Flash VL","provider":"InclusionAI","providerId":"inclusionai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1530,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1530,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":131072,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":12,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":647,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.06,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.18,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":0.18,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/ling-3-0-flash-vl#specifications"},"detailUrl":"/models/ling-3-0-flash-vl"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":53,"rating":1530,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-vl","categoryId":"coding","score":53,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash VL scores 25, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-vl"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-vl","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash VL scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-vl"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-vl","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash VL scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-vl"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-vl","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash VL scores 25, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-vl"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-vl","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash VL scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-vl"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-vl","categoryId":"vision_understanding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash VL scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-vl"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1588,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1530,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1600,"weight":0.22727272727272727},{"categoryId":"writing","rating":1600,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1600,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1600,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1530,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1588,"metrics":{"context_window":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":32768,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.012,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.06,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":647,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.18,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":12,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-vl","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"ling-3-0-flash-fin","name":"Ling 3.0 Flash Fin","provider":"InclusionAI","providerId":"inclusionai","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1520,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1520,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":179,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":623,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.06,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.18,"unit":"$/1M tok","kind":"cost"}],"speedClass":"100–200 tok/s","price":{"value":0.18,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/ling-3-0-flash-fin#specifications"},"detailUrl":"/models/ling-3-0-flash-fin"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":52,"rating":1520,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-fin","categoryId":"coding","score":52,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash Fin scores 23, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-fin"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-fin","categoryId":"reasoning","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash Fin scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-fin"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-fin","categoryId":"writing","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash Fin scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-fin"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-fin","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash Fin scores 23, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-fin"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash-fin","categoryId":"chat","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash Fin scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash-fin"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1578,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1520,"weight":0.25},{"categoryId":"reasoning","rating":1590,"weight":0.25},{"categoryId":"writing","rating":1590,"weight":0.15},{"categoryId":"agents","rating":1610,"weight":0.15},{"categoryId":"chat","rating":1590,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1520,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"100–200 tok/s","latencyClass":"500ms–2s"},"overallElo":1578,"metrics":{"input_price_per_m":{"value":0.06,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":623,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.012,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":235929,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":179,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.18,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash-fin","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"glm-5-3-flash","name":"GLM 5.3 Flash","provider":"Z.ai","providerId":"zai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1600,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":35,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4172,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.15,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":0.5,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/glm-5-3-flash#specifications"},"detailUrl":"/models/glm-5-3-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"glm-5-3-flash","categoryId":"coding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.3 Flash scores 42, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"glm-5-3-flash","categoryId":"reasoning","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.3 Flash scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"glm-5-3-flash","categoryId":"writing","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.3 Flash scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"glm-5-3-flash","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.3 Flash scores 42, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"glm-5-3-flash","categoryId":"chat","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.3 Flash scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"glm-5-3-flash","categoryId":"vision_understanding","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.3 Flash scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-3-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1645,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1600,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1660,"weight":0.22727272727272727},{"categoryId":"writing","rating":1660,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1660,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1660,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1600,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1645,"metrics":{"ttft_ms":{"value":4172,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":35,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.03,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.3-flash","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"qwen3-8-27b","name":"Qwen3.8 27B","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1570,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1570,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":37,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3048,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":2.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":2.5,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/qwen3-8-27b#specifications"},"detailUrl":"/models/qwen3-8-27b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":57,"rating":1570,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-8-27b","categoryId":"coding","score":57,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 27B scores 34, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-8-27b","categoryId":"reasoning","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 27B scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-8-27b","categoryId":"writing","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 27B scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-8-27b","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 27B scores 34, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-8-27b","categoryId":"chat","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 27B scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"qwen3-8-27b","categoryId":"vision_understanding","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 27B scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1618,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1570,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1630,"weight":0.22727272727272727},{"categoryId":"writing","rating":1630,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1630,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1630,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1570,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1618,"metrics":{"input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":2.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":3048,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":37,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.05,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":235929,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-27b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"qwen3-8-2-4t-a95b","name":"Qwen3.8 2.4T A95B","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":30,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1410,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":6,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":6,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/qwen3-8-2-4t-a95b#specifications"},"detailUrl":"/models/qwen3-8-2-4t-a95b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-8-2-4t-a95b","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 2.4T A95B scores 40, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-2-4t-a95b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-8-2-4t-a95b","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 2.4T A95B scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-2-4t-a95b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-8-2-4t-a95b","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 2.4T A95B scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-2-4t-a95b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-8-2-4t-a95b","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 2.4T A95B scores 40, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-2-4t-a95b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-8-2-4t-a95b","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.8 2.4T A95B scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-8-2-4t-a95b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1634,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.25},{"categoryId":"reasoning","rating":1650,"weight":0.25},{"categoryId":"writing","rating":1650,"weight":0.15},{"categoryId":"agents","rating":1630,"weight":0.15},{"categoryId":"chat","rating":1650,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1634,"metrics":{"tokens_per_sec":{"value":30,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":6,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1410,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.8-2.4t-a95b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"nemotron-3-5-lightning","name":"Nemotron 3.5 Lightning","provider":"NVIDIA","providerId":"nvidia","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1490,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1490,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":55,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1785,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.08,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":0.2,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/nemotron-3-5-lightning#specifications"},"detailUrl":"/models/nemotron-3-5-lightning"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":49,"rating":1490,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"nemotron-3-5-lightning","categoryId":"coding","score":49,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3.5 Lightning scores 14, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nemotron-3-5-lightning"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"nemotron-3-5-lightning","categoryId":"reasoning","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3.5 Lightning scores 14, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nemotron-3-5-lightning"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"nemotron-3-5-lightning","categoryId":"writing","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3.5 Lightning scores 14, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nemotron-3-5-lightning"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"nemotron-3-5-lightning","categoryId":"agents","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3.5 Lightning scores 14, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nemotron-3-5-lightning"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"nemotron-3-5-lightning","categoryId":"chat","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3.5 Lightning scores 14, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nemotron-3-5-lightning"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1552,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1490,"weight":0.25},{"categoryId":"reasoning","rating":1560,"weight":0.25},{"categoryId":"writing","rating":1560,"weight":0.15},{"categoryId":"agents","rating":1600,"weight":0.15},{"categoryId":"chat","rating":1560,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1490,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1552,"metrics":{"tokens_per_sec":{"value":55,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":235929,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1785,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.08,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.04,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3.5-lightning","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"solar-pro4","name":"Solar Pro 4","provider":"Upstage","providerId":"upstage","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1540,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1540,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":524288,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":36,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2886,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.09,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.36,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":0.36,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/solar-pro4#specifications"},"detailUrl":"/models/solar-pro4"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":54,"rating":1540,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"solar-pro4","categoryId":"coding","score":54,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Solar Pro 4 scores 28, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/solar-pro4"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"solar-pro4","categoryId":"reasoning","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Solar Pro 4 scores 28, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/solar-pro4"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"solar-pro4","categoryId":"writing","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Solar Pro 4 scores 28, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/solar-pro4"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"solar-pro4","categoryId":"agents","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Solar Pro 4 scores 28, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/solar-pro4"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"solar-pro4","categoryId":"chat","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Solar Pro 4 scores 28, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/solar-pro4"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1597,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1540,"weight":0.25},{"categoryId":"reasoning","rating":1610,"weight":0.25},{"categoryId":"writing","rating":1610,"weight":0.15},{"categoryId":"agents","rating":1620,"weight":0.15},{"categoryId":"chat","rating":1610,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1540,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1597,"metrics":{"tokens_per_sec":{"value":36,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2886,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.09,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":524288,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.36,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.018,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/upstage/solar-pro4","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"muse-spark-1-2","name":"Muse Spark 1.2","provider":"Meta","providerId":"meta","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":100,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4308,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.25,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.25,"unit":"$/1M tok","kind":"cost"}],"speedClass":"100–200 tok/s","price":{"value":4.25,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/muse-spark-1-2#specifications"},"detailUrl":"/models/muse-spark-1-2"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"muse-spark-1-2","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.2 scores 40, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"muse-spark-1-2","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.2 scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-spark-1-2","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.2 scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"muse-spark-1-2","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.2 scores 40, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-spark-1-2","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.2 scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"muse-spark-1-2","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.2 scores 40, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1635,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"100–200 tok/s","latencyClass":"2s+"},"overallElo":1635,"metrics":{"output_price_per_m":{"value":4.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":4308,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":943718,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":100,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.2","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"inkling-small","name":"Inkling Small","provider":"Thinking Machines","providerId":"thinkingmachines","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1540,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1540,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":524288,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":61,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":422,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.45,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":1.2,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/inkling-small#specifications"},"detailUrl":"/models/inkling-small"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":54,"rating":1540,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"inkling-small","categoryId":"coding","score":54,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling Small scores 26, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling-small"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"inkling-small","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling Small scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling-small"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"inkling-small","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling Small scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling-small"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"inkling-small","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling Small scores 26, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling-small"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"inkling-small","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling Small scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling-small"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"inkling-small","categoryId":"vision_understanding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling Small scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling-small"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1590,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1540,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1600,"weight":0.22727272727272727},{"categoryId":"writing","rating":1600,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1600,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1600,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1540,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"200–500ms"},"overallElo":1590,"metrics":{"context_window":{"value":524288,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":422,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.45,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":61,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling-small","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"claude-opus-5","name":"Claude Opus 5","provider":"Anthropic","providerId":"anthropic","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1630,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":59,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3652,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":5,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":25,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":25,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/claude-opus-5#specifications"},"detailUrl":"/models/claude-opus-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"claude-opus-5","categoryId":"coding","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5 scores 51, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"claude-opus-5","categoryId":"reasoning","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5 scores 51, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-opus-5","categoryId":"writing","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5 scores 51, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"claude-opus-5","categoryId":"agents","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5 scores 51, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-opus-5","categoryId":"chat","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5 scores 51, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"claude-opus-5","categoryId":"vision_understanding","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5 scores 51, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1673,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1630,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1690,"weight":0.22727272727272727},{"categoryId":"writing","rating":1690,"weight":0.13636363636363635},{"categoryId":"agents","rating":1650,"weight":0.13636363636363635},{"categoryId":"chat","rating":1690,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1690,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1630,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1673,"metrics":{"tokens_per_sec":{"value":59,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":3652,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"ling-3-0-flash","name":"Ling 3.0 Flash","provider":"InclusionAI","providerId":"inclusionai","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1520,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1520,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":34,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":773,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.021,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.063,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":0.063,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/ling-3-0-flash#specifications"},"detailUrl":"/models/ling-3-0-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":52,"rating":1520,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash","categoryId":"coding","score":52,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash scores 21, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash","categoryId":"reasoning","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash scores 21, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash","categoryId":"writing","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash scores 21, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash scores 21, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"ling-3-0-flash","categoryId":"chat","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Ling 3.0 Flash scores 21, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/ling-3-0-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1572,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1520,"weight":0.25},{"categoryId":"reasoning","rating":1580,"weight":0.25},{"categoryId":"writing","rating":1580,"weight":0.15},{"categoryId":"agents","rating":1610,"weight":0.15},{"categoryId":"chat","rating":1580,"weight":0.2}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1520,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1572,"metrics":{"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":773,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":34,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":32768,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.063,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.0042,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.021,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/inclusionai/ling-3.0-flash","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","provider":"Google","providerId":"google","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1570,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1570,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":53,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1290,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.75,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":3.75,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":3.75,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gemini-3-6-flash#specifications"},"detailUrl":"/models/gemini-3-6-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":57,"rating":1570,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gemini-3-6-flash","categoryId":"coding","score":57,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.6 Flash scores 34, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-6-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gemini-3-6-flash","categoryId":"reasoning","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.6 Flash scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-6-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-6-flash","categoryId":"writing","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.6 Flash scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-6-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gemini-3-6-flash","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.6 Flash scores 34, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-6-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-6-flash","categoryId":"chat","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.6 Flash scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-6-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gemini-3-6-flash","categoryId":"vision_understanding","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.6 Flash scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-6-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1618,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1570,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1630,"weight":0.22727272727272727},{"categoryId":"writing","rating":1630,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1630,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1630,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1570,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1618,"metrics":{"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.75,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1290,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.075,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":3.75,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":53,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.6-flash","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash Lite","provider":"Google","providerId":"google","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1520,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1520,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":66,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":599,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":2.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":2.5,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gemini-3-5-flash-lite#specifications"},"detailUrl":"/models/gemini-3-5-flash-lite"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":52,"rating":1520,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash-lite","categoryId":"coding","score":52,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash Lite scores 23, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash-lite"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash-lite","categoryId":"reasoning","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash Lite scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash-lite"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash-lite","categoryId":"writing","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash Lite scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash-lite"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash-lite","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash Lite scores 23, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash-lite"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash-lite","categoryId":"chat","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash Lite scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash-lite"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash-lite","categoryId":"vision_understanding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash Lite scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash-lite"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1579,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1520,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1590,"weight":0.22727272727272727},{"categoryId":"writing","rating":1590,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1590,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1590,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1520,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1579,"metrics":{"max_output_tokens":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":66,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":2.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":599,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.03,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash-lite","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"longcat-2-0","name":"LongCat 2.0","provider":"Meituan","providerId":"meituan","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1510,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1510,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048756,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":37,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2985,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":1.2,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/longcat-2-0#specifications"},"detailUrl":"/models/longcat-2-0"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":51,"rating":1510,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"longcat-2-0","categoryId":"coding","score":51,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which LongCat 2.0 scores 20, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/longcat-2-0"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"longcat-2-0","categoryId":"reasoning","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which LongCat 2.0 scores 20, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/longcat-2-0"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"longcat-2-0","categoryId":"writing","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which LongCat 2.0 scores 20, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/longcat-2-0"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"longcat-2-0","categoryId":"agents","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which LongCat 2.0 scores 20, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/longcat-2-0"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"longcat-2-0","categoryId":"chat","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which LongCat 2.0 scores 20, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/longcat-2-0"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1568,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1510,"weight":0.25},{"categoryId":"reasoning","rating":1580,"weight":0.25},{"categoryId":"writing","rating":1580,"weight":0.15},{"categoryId":"agents","rating":1600,"weight":0.15},{"categoryId":"chat","rating":1580,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1510,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1568,"metrics":{"context_window":{"value":1048756,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2985,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.006,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":37,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meituan/longcat-2.0","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"inkling","name":"Inkling","provider":"Thinking Machines","providerId":"thinkingmachines","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1540,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1540,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":34,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":594,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.05,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":4.05,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/inkling#specifications"},"detailUrl":"/models/inkling"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":54,"rating":1540,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"inkling","categoryId":"coding","score":54,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling scores 26, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"inkling","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"inkling","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"inkling","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling scores 26, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"inkling","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"inkling","categoryId":"vision_understanding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Inkling scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/inkling"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1590,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1540,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1600,"weight":0.22727272727272727},{"categoryId":"writing","rating":1600,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1600,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1600,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1540,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1590,"metrics":{"max_output_tokens":{"value":32768,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.17,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":4.05,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":594,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":34,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/thinkingmachines/inkling","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"kimi-k3","name":"Kimi K3","provider":"Moonshot AI","providerId":"moonshotai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1610,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1670,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":20,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":6382,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":15,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":15,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/kimi-k3#specifications"},"detailUrl":"/models/kimi-k3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"kimi-k3","categoryId":"coding","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K3 scores 44, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"kimi-k3","categoryId":"reasoning","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K3 scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"kimi-k3","categoryId":"writing","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K3 scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"kimi-k3","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K3 scores 44, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"kimi-k3","categoryId":"chat","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K3 scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":67,"rating":1670,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"kimi-k3","categoryId":"vision_understanding","score":67,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K3 scores 44, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1654,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1610,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1670,"weight":0.22727272727272727},{"categoryId":"writing","rating":1670,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1670,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1670,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1610,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1654,"metrics":{"input_price_per_m":{"value":3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":20,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":943718,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":6382,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k3","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"muse-spark-1-1","name":"Muse Spark 1.1","provider":"Meta","providerId":"meta","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1570,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1570,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":233,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2806,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.25,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.25,"unit":"$/1M tok","kind":"cost"}],"speedClass":"200+ tok/s","price":{"value":4.25,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/muse-spark-1-1#specifications"},"detailUrl":"/models/muse-spark-1-1"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":57,"rating":1570,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"muse-spark-1-1","categoryId":"coding","score":57,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.1 scores 34, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"muse-spark-1-1","categoryId":"reasoning","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.1 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-spark-1-1","categoryId":"writing","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.1 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"muse-spark-1-1","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.1 scores 34, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"muse-spark-1-1","categoryId":"chat","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.1 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"muse-spark-1-1","categoryId":"vision_understanding","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Muse Spark 1.1 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/muse-spark-1-1"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1618,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1570,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1630,"weight":0.22727272727272727},{"categoryId":"writing","rating":1630,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1630,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1630,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1570,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"200+ tok/s","latencyClass":"2s+"},"overallElo":1618,"metrics":{"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":4.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":233,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":943718,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2806,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-spark-1.1","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["file","image","text"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1580,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":55,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2215,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":1.2,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gpt-5-6-luna#specifications"},"detailUrl":"/models/gpt-5-6-luna"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gpt-5-6-luna","categoryId":"coding","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Luna scores 38, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gpt-5-6-luna","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Luna scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-6-luna","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Luna scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gpt-5-6-luna","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Luna scores 38, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-6-luna","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Luna scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gpt-5-6-luna","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Luna scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1634,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1580,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1580,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1634,"metrics":{"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":55,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2215,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.02,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-luna","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["file","image","text"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1600,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":41,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3128,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":12,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":12,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gpt-5-6-terra#specifications"},"detailUrl":"/models/gpt-5-6-terra"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gpt-5-6-terra","categoryId":"coding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Terra scores 42, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-terra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gpt-5-6-terra","categoryId":"reasoning","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Terra scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-terra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-6-terra","categoryId":"writing","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Terra scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-terra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gpt-5-6-terra","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Terra scores 42, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-terra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-6-terra","categoryId":"chat","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Terra scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-terra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gpt-5-6-terra","categoryId":"vision_understanding","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Terra scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-terra"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1645,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1600,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1660,"weight":0.22727272727272727},{"categoryId":"writing","rating":1660,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1660,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1660,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1600,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1645,"metrics":{"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":12,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":3128,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":41,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-terra","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["file","image","text"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1620,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":35,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4621,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":10,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":10,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gpt-5-6-sol#specifications"},"detailUrl":"/models/gpt-5-6-sol"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gpt-5-6-sol","categoryId":"coding","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Sol scores 47, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gpt-5-6-sol","categoryId":"reasoning","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Sol scores 47, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-6-sol","categoryId":"writing","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Sol scores 47, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gpt-5-6-sol","categoryId":"agents","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Sol scores 47, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-6-sol","categoryId":"chat","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Sol scores 47, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gpt-5-6-sol","categoryId":"vision_understanding","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.6 Sol scores 47, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1664,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1620,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1680,"weight":0.22727272727272727},{"categoryId":"writing","rating":1680,"weight":0.13636363636363635},{"categoryId":"agents","rating":1650,"weight":0.13636363636363635},{"categoryId":"chat","rating":1680,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1680,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1620,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1664,"metrics":{"tokens_per_sec":{"value":35,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":10,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":4621,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.6-sol","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"grok-4-5","name":"Grok 4.5","provider":"xAI","providerId":"xai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":500000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":47,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1412,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":6,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":6,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/grok-4-5#specifications"},"detailUrl":"/models/grok-4-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"grok-4-5","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.5 scores 39, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"grok-4-5","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-5","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"grok-4-5","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.5 scores 39, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-5","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"grok-4-5","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1635,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1635,"metrics":{"cached_input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":47,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1412,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":450000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":6,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":500000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.5","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"hy3","name":"Hy3","provider":"Tencent","providerId":"tencent","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1540,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1540,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":56,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2673,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.132,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.528,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":0.528,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/hy3#specifications"},"detailUrl":"/models/hy3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":54,"rating":1540,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"hy3","categoryId":"coding","score":54,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Hy3 scores 26, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/hy3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"hy3","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Hy3 scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/hy3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"hy3","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Hy3 scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/hy3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"hy3","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Hy3 scores 26, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/hy3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"hy3","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Hy3 scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/hy3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1588,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1540,"weight":0.25},{"categoryId":"reasoning","rating":1600,"weight":0.25},{"categoryId":"writing","rating":1600,"weight":0.15},{"categoryId":"agents","rating":1610,"weight":0.15},{"categoryId":"chat","rating":1600,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1540,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1588,"metrics":{"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":0.528,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2673,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.033,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.132,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":56,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/tencent/hy3","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"claude-sonnet-5","name":"Claude Sonnet 5","provider":"Anthropic","providerId":"anthropic","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1580,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":49,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2193,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":10,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":10,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/claude-sonnet-5#specifications"},"detailUrl":"/models/claude-sonnet-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"claude-sonnet-5","categoryId":"coding","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Sonnet 5 scores 38, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-sonnet-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"claude-sonnet-5","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Sonnet 5 scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-sonnet-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-sonnet-5","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Sonnet 5 scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-sonnet-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"claude-sonnet-5","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Sonnet 5 scores 38, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-sonnet-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-sonnet-5","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Sonnet 5 scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-sonnet-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"claude-sonnet-5","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Sonnet 5 scores 38, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-sonnet-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1634,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1580,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1580,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1634,"metrics":{"output_price_per_m":{"value":10,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":49,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2193,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-sonnet-5","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"glm-5-2","name":"GLM 5.2","provider":"Z.ai","providerId":"zai","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1570,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1570,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":49,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3137,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.4,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.4,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":4.4,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/glm-5-2#specifications"},"detailUrl":"/models/glm-5-2"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":57,"rating":1570,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"glm-5-2","categoryId":"coding","score":57,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.2 scores 34, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"glm-5-2","categoryId":"reasoning","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.2 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"glm-5-2","categoryId":"writing","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.2 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"glm-5-2","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.2 scores 34, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"glm-5-2","categoryId":"chat","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GLM 5.2 scores 34, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/glm-5-2"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1617,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1570,"weight":0.25},{"categoryId":"reasoning","rating":1630,"weight":0.25},{"categoryId":"writing","rating":1630,"weight":0.15},{"categoryId":"agents","rating":1630,"weight":0.15},{"categoryId":"chat","rating":1630,"weight":0.2}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1570,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1617,"metrics":{"ttft_ms":{"value":3137,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.4,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.26,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":4.4,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":49,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/z-ai/glm-5.2","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","provider":"Moonshot AI","providerId":"moonshotai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1540,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1540,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":114,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2670,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.9,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":8,"unit":"$/1M tok","kind":"cost"}],"speedClass":"100–200 tok/s","price":{"value":8,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/kimi-k2-7-code#specifications"},"detailUrl":"/models/kimi-k2-7-code"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":54,"rating":1540,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"kimi-k2-7-code","categoryId":"coding","score":54,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K2.7 Code scores 26, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k2-7-code"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"kimi-k2-7-code","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K2.7 Code scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k2-7-code"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"kimi-k2-7-code","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K2.7 Code scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k2-7-code"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"kimi-k2-7-code","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K2.7 Code scores 26, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k2-7-code"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"kimi-k2-7-code","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K2.7 Code scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k2-7-code"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"kimi-k2-7-code","categoryId":"vision_understanding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Kimi K2.7 Code scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/kimi-k2-7-code"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1590,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1540,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1600,"weight":0.22727272727272727},{"categoryId":"writing","rating":1600,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1600,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1600,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1540,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"100–200 tok/s","latencyClass":"2s+"},"overallElo":1590,"metrics":{"max_output_tokens":{"value":235929,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2670,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":114,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":8,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.38,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.9,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/moonshotai/kimi-k2.7-code","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"claude-fable-5","name":"Claude Fable 5","provider":"Anthropic","providerId":"anthropic","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1630,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1690,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":42,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":6090,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":10,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":50,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":50,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/claude-fable-5#specifications"},"detailUrl":"/models/claude-fable-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"claude-fable-5","categoryId":"coding","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5 scores 50, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"claude-fable-5","categoryId":"reasoning","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5 scores 50, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-fable-5","categoryId":"writing","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5 scores 50, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"claude-fable-5","categoryId":"agents","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5 scores 50, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-fable-5","categoryId":"chat","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5 scores 50, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":69,"rating":1690,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"claude-fable-5","categoryId":"vision_understanding","score":69,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Fable 5 scores 50, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-fable-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1673,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1630,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1690,"weight":0.22727272727272727},{"categoryId":"writing","rating":1690,"weight":0.13636363636363635},{"categoryId":"agents","rating":1650,"weight":0.13636363636363635},{"categoryId":"chat","rating":1690,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1690,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1630,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1673,"metrics":{"tokens_per_sec":{"value":42,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":10,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":50,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":6090,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-fable-5","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra","provider":"NVIDIA","providerId":"nvidia","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1520,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1520,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":36,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4769,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.5,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":2.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":2.2,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/nemotron-3-ultra-550b-a55b#specifications"},"detailUrl":"/models/nemotron-3-ultra-550b-a55b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":52,"rating":1520,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"nemotron-3-ultra-550b-a55b","categoryId":"coding","score":52,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3 Ultra scores 23, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nvidia-nemotron-3-ultra-550b-a55b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"nemotron-3-ultra-550b-a55b","categoryId":"reasoning","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3 Ultra scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nvidia-nemotron-3-ultra-550b-a55b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"nemotron-3-ultra-550b-a55b","categoryId":"writing","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3 Ultra scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nvidia-nemotron-3-ultra-550b-a55b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"nemotron-3-ultra-550b-a55b","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3 Ultra scores 23, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nvidia-nemotron-3-ultra-550b-a55b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"nemotron-3-ultra-550b-a55b","categoryId":"chat","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Nemotron 3 Ultra scores 23, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/nvidia-nemotron-3-ultra-550b-a55b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1578,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1520,"weight":0.25},{"categoryId":"reasoning","rating":1590,"weight":0.25},{"categoryId":"writing","rating":1590,"weight":0.15},{"categoryId":"agents","rating":1610,"weight":0.15},{"categoryId":"chat","rating":1590,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1520,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1578,"metrics":{"ttft_ms":{"value":4769,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":36,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":16384,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":2.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/nvidia/nemotron-3-ultra-550b-a55b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1540,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1540,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":12,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":813,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.32,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.28,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":1.28,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/qwen3-7-plus#specifications"},"detailUrl":"/models/qwen3-7-plus"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":54,"rating":1540,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-7-plus","categoryId":"coding","score":54,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Plus scores 26, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-7-plus","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Plus scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-7-plus","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Plus scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-7-plus","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Plus scores 26, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-7-plus","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Plus scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"qwen3-7-plus","categoryId":"vision_understanding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Plus scores 26, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1590,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1540,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1600,"weight":0.22727272727272727},{"categoryId":"writing","rating":1600,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1600,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1600,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1540,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1590,"metrics":{"input_price_per_m":{"value":0.32,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.064,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1.28,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":813,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":12,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-plus","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"minimax-m3","name":"MiniMax M3","provider":"MiniMax","providerId":"minimax","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1550,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":524288,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":84,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":995,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":1.2,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/minimax-m3#specifications"},"detailUrl":"/models/minimax-m3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"minimax-m3","categoryId":"coding","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiniMax M3 scores 30, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/minimax-m3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"minimax-m3","categoryId":"reasoning","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiniMax M3 scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/minimax-m3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"minimax-m3","categoryId":"writing","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiniMax M3 scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/minimax-m3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"minimax-m3","categoryId":"agents","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiniMax M3 scores 30, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/minimax-m3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"minimax-m3","categoryId":"chat","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiniMax M3 scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/minimax-m3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"minimax-m3","categoryId":"vision_understanding","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiniMax M3 scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/minimax-m3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1606,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1550,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1620,"weight":0.22727272727272727},{"categoryId":"writing","rating":1620,"weight":0.13636363636363635},{"categoryId":"agents","rating":1620,"weight":0.13636363636363635},{"categoryId":"chat","rating":1620,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1620,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1550,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1606,"metrics":{"max_output_tokens":{"value":512000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":84,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":524288,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":995,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.06,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/minimax-m3","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"step-3-7-flash","name":"Step 3.7 Flash","provider":"StepFun","providerId":"stepfun","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1510,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1510,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":256000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":61,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":3551,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.15,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":1.15,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/step-3-7-flash#specifications"},"detailUrl":"/models/step-3-7-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":51,"rating":1510,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"step-3-7-flash","categoryId":"coding","score":51,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Step 3.7 Flash scores 19, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/step-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"step-3-7-flash","categoryId":"reasoning","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Step 3.7 Flash scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/step-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"step-3-7-flash","categoryId":"writing","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Step 3.7 Flash scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/step-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"step-3-7-flash","categoryId":"agents","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Step 3.7 Flash scores 19, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/step-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"step-3-7-flash","categoryId":"chat","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Step 3.7 Flash scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/step-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"step-3-7-flash","categoryId":"vision_understanding","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Step 3.7 Flash scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/step-3-7-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1569,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1510,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1580,"weight":0.22727272727272727},{"categoryId":"writing","rating":1580,"weight":0.13636363636363635},{"categoryId":"agents","rating":1600,"weight":0.13636363636363635},{"categoryId":"chat","rating":1580,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1580,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1510,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1569,"metrics":{"cached_input_price_per_m":{"value":0.04,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":61,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":230400,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":3551,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":256000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/stepfun/step-3.7-flash","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"claude-opus-4-8","name":"Claude Opus 4.8","provider":"Anthropic","providerId":"anthropic","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1600,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":61,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1996,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":5,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":25,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":25,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/claude-opus-4-8#specifications"},"detailUrl":"/models/claude-opus-4-8"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"claude-opus-4-8","categoryId":"coding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 4.8 scores 42, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-4-8"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"claude-opus-4-8","categoryId":"reasoning","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 4.8 scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-4-8"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-opus-4-8","categoryId":"writing","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 4.8 scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-4-8"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"claude-opus-4-8","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 4.8 scores 42, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-4-8"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-opus-4-8","categoryId":"chat","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 4.8 scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-4-8"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"claude-opus-4-8","categoryId":"vision_understanding","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 4.8 scores 42, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-4-8"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1645,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1600,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1660,"weight":0.22727272727272727},{"categoryId":"writing","rating":1660,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1660,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1660,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1600,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1645,"metrics":{"output_price_per_m":{"value":25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":61,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1996,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-4.8","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"qwen3-7-max","name":"Qwen3.7 Max","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1550,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":57,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1083,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.475,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.425,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":4.425,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/qwen3-7-max#specifications"},"detailUrl":"/models/qwen3-7-max"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-7-max","categoryId":"coding","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Max scores 30, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-7-max","categoryId":"reasoning","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Max scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-7-max","categoryId":"writing","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Max scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-7-max","categoryId":"agents","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Max scores 30, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-7-max","categoryId":"chat","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.7 Max scores 30, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-7-max"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1605,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1550,"weight":0.25},{"categoryId":"reasoning","rating":1620,"weight":0.25},{"categoryId":"writing","rating":1620,"weight":0.15},{"categoryId":"agents","rating":1620,"weight":0.15},{"categoryId":"chat","rating":1620,"weight":0.2}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1550,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1605,"metrics":{"cached_input_price_per_m":{"value":0.295,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":4.425,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.475,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":57,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1083,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.7-max","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","provider":"Google","providerId":"google","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","file","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1560,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":78,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1332,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.5,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":9,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":9,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gemini-3-5-flash#specifications"},"detailUrl":"/models/gemini-3-5-flash"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash","categoryId":"coding","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash scores 33, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash","categoryId":"reasoning","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash scores 33, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash","categoryId":"writing","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash scores 33, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash","categoryId":"agents","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash scores 33, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash","categoryId":"chat","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash scores 33, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gemini-3-5-flash","categoryId":"vision_understanding","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Gemini 3.5 Flash scores 33, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gemini-3-5-flash"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1615,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1560,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1630,"weight":0.22727272727272727},{"categoryId":"writing","rating":1630,"weight":0.13636363636363635},{"categoryId":"agents","rating":1620,"weight":0.13636363636363635},{"categoryId":"chat","rating":1630,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1630,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1560,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1615,"metrics":{"tokens_per_sec":{"value":78,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":9,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1332,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.5-flash","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"grok-4-3","name":"Grok 4.3","provider":"xAI","providerId":"xai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1530,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1530,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":83,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":582,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.25,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":2.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":2.5,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/grok-4-3#specifications"},"detailUrl":"/models/grok-4-3"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":53,"rating":1530,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"grok-4-3","categoryId":"coding","score":53,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.3 scores 25, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"grok-4-3","categoryId":"reasoning","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.3 scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-3","categoryId":"writing","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.3 scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"grok-4-3","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.3 scores 25, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-3","categoryId":"chat","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.3 scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"grok-4-3","categoryId":"vision_understanding","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.3 scores 25, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-3"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1588,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1530,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1600,"weight":0.22727272727272727},{"categoryId":"writing","rating":1600,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1600,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1600,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1530,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1588,"metrics":{"cached_input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":900000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.25,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":582,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":2.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":83,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.3","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"mistral-medium-3-5","name":"Mistral Medium 3.5","provider":"Mistral","providerId":"mistralai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1490,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1490,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1560,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":37,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1237,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.5,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":7.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":7.5,"unit":"USD/1M output tokens"},"specifications":{"count":9,"published":8,"detailUrl":"/models/mistral-medium-3-5#specifications"},"detailUrl":"/models/mistral-medium-3-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":49,"rating":1490,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"mistral-medium-3-5","categoryId":"coding","score":49,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Mistral Medium 3.5 scores 15, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mistral-medium-3-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"mistral-medium-3-5","categoryId":"reasoning","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Mistral Medium 3.5 scores 15, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mistral-medium-3-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"mistral-medium-3-5","categoryId":"writing","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Mistral Medium 3.5 scores 15, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mistral-medium-3-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"mistral-medium-3-5","categoryId":"agents","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Mistral Medium 3.5 scores 15, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mistral-medium-3-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"mistral-medium-3-5","categoryId":"chat","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Mistral Medium 3.5 scores 15, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mistral-medium-3-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":56,"rating":1560,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"mistral-medium-3-5","categoryId":"vision_understanding","score":56,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Mistral Medium 3.5 scores 15, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mistral-medium-3-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1553,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1490,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1560,"weight":0.22727272727272727},{"categoryId":"writing","rating":1560,"weight":0.13636363636363635},{"categoryId":"agents","rating":1600,"weight":0.13636363636363635},{"categoryId":"chat","rating":1560,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1560,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1490,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1553,"metrics":{"tokens_per_sec":{"value":37,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":209715,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":7.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":1.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1237,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/mistralai/mistral-medium-3-5","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"qwen3-6-35b-a3b","name":"Qwen3.6 35B A3B","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1510,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1510,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1600,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":256000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":57,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":819,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.1,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":1,"unit":"USD/1M output tokens"},"specifications":{"count":9,"published":8,"detailUrl":"/models/qwen3-6-35b-a3b#specifications"},"detailUrl":"/models/qwen3-6-35b-a3b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":51,"rating":1510,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-6-35b-a3b","categoryId":"coding","score":51,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 35B A3B scores 19, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-35b-a3b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-6-35b-a3b","categoryId":"reasoning","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 35B A3B scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-35b-a3b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-6-35b-a3b","categoryId":"writing","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 35B A3B scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-35b-a3b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":60,"rating":1600,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-6-35b-a3b","categoryId":"agents","score":60,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 35B A3B scores 19, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-35b-a3b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-6-35b-a3b","categoryId":"chat","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 35B A3B scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-35b-a3b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"qwen3-6-35b-a3b","categoryId":"vision_understanding","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 35B A3B scores 19, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-35b-a3b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1569,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1510,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1580,"weight":0.22727272727272727},{"categoryId":"writing","rating":1580,"weight":0.13636363636363635},{"categoryId":"agents","rating":1600,"weight":0.13636363636363635},{"categoryId":"chat","rating":1580,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1580,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1510,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1569,"metrics":{"max_output_tokens":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":57,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":0.1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":819,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":256000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-35b-a3b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"qwen3-6-27b","name":"Qwen3.6 27B","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1520,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1520,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":262144,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":21,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1973,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":3.2,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":3.2,"unit":"USD/1M output tokens"},"specifications":{"count":9,"published":8,"detailUrl":"/models/qwen3-6-27b#specifications"},"detailUrl":"/models/qwen3-6-27b"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":52,"rating":1520,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"qwen3-6-27b","categoryId":"coding","score":52,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 27B scores 22, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"qwen3-6-27b","categoryId":"reasoning","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 27B scores 22, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-6-27b","categoryId":"writing","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 27B scores 22, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"qwen3-6-27b","categoryId":"agents","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 27B scores 22, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"qwen3-6-27b","categoryId":"chat","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 27B scores 22, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"qwen3-6-27b","categoryId":"vision_understanding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Qwen3.6 27B scores 22, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/qwen3-6-27b"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1579,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1520,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1590,"weight":0.22727272727272727},{"categoryId":"writing","rating":1590,"weight":0.13636363636363635},{"categoryId":"agents","rating":1610,"weight":0.13636363636363635},{"categoryId":"chat","rating":1590,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1590,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1520,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"500ms–2s"},"overallElo":1579,"metrics":{"input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":262144,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":21,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":235929,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":1973,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":3.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen3.6-27b","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"gpt-5-5","name":"GPT-5.5","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["file","image","text"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1590,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":45,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2728,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":5,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":30,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":30,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gpt-5-5#specifications"},"detailUrl":"/models/gpt-5-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gpt-5-5","categoryId":"coding","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.5 scores 39, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gpt-5-5","categoryId":"reasoning","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-5","categoryId":"writing","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gpt-5-5","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.5 scores 39, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-5-5","categoryId":"chat","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gpt-5-5","categoryId":"vision_understanding","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-5.5 scores 39, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1635,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1590,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1650,"weight":0.22727272727272727},{"categoryId":"writing","rating":1650,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1650,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1650,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1590,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1635,"metrics":{"tokens_per_sec":{"value":45,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"},"input_price_per_m":{"value":5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":30,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2728,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-5.5","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro 0423","provider":"DeepSeek","providerId":"deepseek","presentation":{"task":"text","taskLabel":"Text generation","modality":"text","capabilities":{"inputs":["text"],"outputs":["text"],"hasVision":false,"openWeights":false},"headline":{"value":1580,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":38,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2443,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.50162,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":3.135,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":3.135,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/deepseek-v4-pro#specifications"},"detailUrl":"/models/deepseek-v4-pro"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"deepseek-v4-pro","categoryId":"coding","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4 Pro 0423 scores 36, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"deepseek-v4-pro","categoryId":"reasoning","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4 Pro 0423 scores 36, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"deepseek-v4-pro","categoryId":"writing","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4 Pro 0423 scores 36, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"deepseek-v4-pro","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4 Pro 0423 scores 36, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"deepseek-v4-pro","categoryId":"chat","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which DeepSeek V4 Pro 0423 scores 36, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/deepseek-v4-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1625,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1580,"weight":0.25},{"categoryId":"reasoning","rating":1640,"weight":0.25},{"categoryId":"writing","rating":1640,"weight":0.15},{"categoryId":"agents","rating":1630,"weight":0.15},{"categoryId":"chat","rating":1640,"weight":0.2}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1580,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1625,"metrics":{"input_price_per_m":{"value":1.50162,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"},"output_price_per_m":{"value":3.135,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"},"cached_input_price_per_m":{"value":0.135,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"},"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"},"ttft_ms":{"value":2443,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"},"tokens_per_sec":{"value":38,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"},"max_output_tokens":{"value":393216,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/deepseek/deepseek-v4-pro","verificationStatus":"verified","observedAt":"2026-09-16"}}},{"id":"grok-imagine-image-2-0","name":"Grok Imagine Image 2.0","provider":"xAI","providerId":"xai","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["text","image"],"outputs":["image"],"hasVision":false,"openWeights":false},"headline":{"value":1830,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1830,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":12.72,"unit":"s","kind":"speed"}],"speedClass":"5–15s","price":null,"specifications":{"count":7,"published":7,"detailUrl":"/models/grok-imagine-image-2-0#specifications"},"detailUrl":"/models/grok-imagine-image-2-0"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":83,"rating":1830,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"grok-imagine-image-2-0","categoryId":"image_generation","score":83,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Grok Imagine Image 2.0 scores 1153.67. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1830,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1830,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"image_generation":{"elo":1830,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1830,"metrics":{"max_output_tokens":{"value":58982,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-imagine-image-2.0","verificationStatus":"verified","observedAt":"2026-09-21"},"context_window":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-imagine-image-2.0","verificationStatus":"verified","observedAt":"2026-09-21"},"image_gen_time_sec":{"value":12.72,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-imagine-image-2.0","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"mai-image-2-6","name":"MAI-Image-2.6","provider":"Microsoft AI","providerId":"microsoft","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["text","image"],"outputs":["image"],"hasVision":false,"openWeights":false},"headline":{"value":1820,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1820,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":25.08,"unit":"s","kind":"speed"}],"speedClass":"15–30s","price":null,"specifications":{"count":7,"published":7,"detailUrl":"/models/mai-image-2-6#specifications"},"detailUrl":"/models/mai-image-2-6"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":82,"rating":1820,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"mai-image-2-6","categoryId":"image_generation","score":82,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where MAI-Image-2.6 scores 1147.14. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1820,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1820,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"image_generation":{"elo":1820,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1820,"metrics":{"input_price_per_m":{"value":5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/microsoft/mai-image-2.6","verificationStatus":"verified","observedAt":"2026-09-21"},"context_window":{"value":4096,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/microsoft/mai-image-2.6","verificationStatus":"verified","observedAt":"2026-09-21"},"max_output_tokens":{"value":1024,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/microsoft/mai-image-2.6","verificationStatus":"verified","observedAt":"2026-09-21"},"image_gen_time_sec":{"value":25.08,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/microsoft/mai-image-2.6","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"gemini-3-1-flash-image","name":"Nano Banana 2 (Gemini 3.1 Flash Image)","provider":"Google","providerId":"google","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["image","text"],"outputs":["image","text"],"hasVision":false,"openWeights":false},"headline":{"value":1780,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1780,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":17.69,"unit":"s","kind":"speed"}],"speedClass":"15–30s","price":null,"specifications":{"count":8,"published":8,"detailUrl":"/models/gemini-3-1-flash-image#specifications"},"detailUrl":"/models/gemini-3-1-flash-image"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":78,"rating":1780,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"gemini-3-1-flash-image","categoryId":"image_generation","score":78,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Nano Banana 2 (Gemini 3.1 Flash Image) scores 1122. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1780,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1780,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"image_generation":{"elo":1780,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1780,"metrics":{"context_window":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image","verificationStatus":"verified","observedAt":"2026-09-21"},"output_price_per_m":{"value":3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image","verificationStatus":"verified","observedAt":"2026-09-21"},"input_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image","verificationStatus":"verified","observedAt":"2026-09-21"},"max_output_tokens":{"value":58982,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image","verificationStatus":"verified","observedAt":"2026-09-21"},"image_gen_time_sec":{"value":17.69,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/gemini-3.1-flash-image","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"muse-image","name":"Muse Image","provider":"Meta","providerId":"meta","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["text","image"],"outputs":["image"],"hasVision":false,"openWeights":false},"headline":{"value":1760,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1760,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":19.42,"unit":"s","kind":"speed"}],"speedClass":"15–30s","price":null,"specifications":{"count":6,"published":5,"detailUrl":"/models/muse-image#specifications"},"detailUrl":"/models/muse-image"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":76,"rating":1760,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"muse-image","categoryId":"image_generation","score":76,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Muse Image scores 1111.42. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1760,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1760,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"image_generation":{"elo":1760,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1760,"metrics":{"max_output_tokens":{"value":58982,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-image","verificationStatus":"verified","observedAt":"2026-09-21"},"context_window":{"value":65536,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-image","verificationStatus":"verified","observedAt":"2026-09-21"},"image_gen_time_sec":{"value":19.42,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/meta/muse-image","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"flux-2-flex","name":"FLUX.2 Flex","provider":"Black Forest Labs","providerId":"black-forest-labs","presentation":{"task":"image_generation","taskLabel":"Image generation","modality":"image","capabilities":{"inputs":["text","image"],"outputs":["image"],"hasVision":false,"openWeights":false},"headline":{"value":1620,"label":"Images Estimated rating","shortLabel":"Images rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"image_generation","isOverall":false},"categories":[{"categoryId":"image_generation","label":"Image generation","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0}],"facts":[{"metric":"image_gen_time_sec","label":"Gen Time","value":13.27,"unit":"s","kind":"speed"}],"speedClass":"5–15s","price":null,"specifications":{"count":6,"published":5,"detailUrl":"/models/flux-2-flex#specifications"},"detailUrl":"/models/flux-2-flex"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"flux-2-flex","categoryId":"image_generation","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where FLUX.2 [flex] scores 1025. Placed by where that sits between HiDream-O1-Image-1.5 at 1022 and GPT Image 2.5 Flare at 1188, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/image/leaderboard/text-to-image"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1620,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"image_generation","rating":1620,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"image_generation":{"elo":1620,"basisLine":"Estimated · Image generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1620,"metrics":{"context_window":{"value":67344,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/black-forest-labs/flux.2-flex","verificationStatus":"verified","observedAt":"2026-09-21"},"max_output_tokens":{"value":60609,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/black-forest-labs/flux.2-flex","verificationStatus":"verified","observedAt":"2026-09-21"},"image_gen_time_sec":{"value":13.27,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/black-forest-labs/flux.2-flex","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"wan-3-0","name":"Wan 3.0","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"video_generation","taskLabel":"Video generation","modality":"video","capabilities":{"inputs":["text","image"],"outputs":["video"],"hasVision":false,"openWeights":false},"headline":{"value":1880,"label":"Video Estimated rating","shortLabel":"Video rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"video_generation","isOverall":false},"categories":[{"categoryId":"video_generation","label":"Video generation","state":"rated","sourceLabel":"Estimated","elo":1880,"coveragePct":0}],"facts":[{"metric":"video_gen_time_sec","label":"Gen Time","value":239.37,"unit":"s","kind":"speed"}],"speedClass":"3min+","price":null,"specifications":{"count":7,"published":6,"detailUrl":"/models/wan-3-0#specifications"},"detailUrl":"/models/wan-3-0"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"rated","score":88,"rating":1880,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"wan-3-0","categoryId":"video_generation","score":88,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Wan 3.0 scores 1229 with a 95% confidence interval of plus or minus 9 over 6,011 blind pairwise votes. Placed by where that sits between Seedance 1.5 pro at 1000 and Gemini Omni Flash at 1233, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/video/leaderboard/text-to-video"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1880,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"video_generation","rating":1880,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"video_generation":{"elo":1880,"basisLine":"Estimated · Video generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1880,"metrics":{"context_window":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/alibaba/wan-3.0","verificationStatus":"verified","observedAt":"2026-09-21"},"max_output_tokens":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/alibaba/wan-3.0","verificationStatus":"verified","observedAt":"2026-09-21"},"video_gen_time_sec":{"value":239.37,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/alibaba/wan-3.0","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"veo-3-1","name":"Veo 3.1","provider":"Google","providerId":"google","presentation":{"task":"video_generation","taskLabel":"Video generation","modality":"video","capabilities":{"inputs":["text","image"],"outputs":["video"],"hasVision":false,"openWeights":false},"headline":{"value":1720,"label":"Video Estimated rating","shortLabel":"Video rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"video_generation","isOverall":false},"categories":[{"categoryId":"video_generation","label":"Video generation","state":"rated","sourceLabel":"Estimated","elo":1720,"coveragePct":0}],"facts":[{"metric":"video_gen_time_sec","label":"Gen Time","value":42.62,"unit":"s","kind":"speed"}],"speedClass":"30–60s","price":null,"specifications":{"count":7,"published":7,"detailUrl":"/models/veo-3-1#specifications"},"detailUrl":"/models/veo-3-1"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"rated","score":72,"rating":1720,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"veo-3-1","categoryId":"video_generation","score":72,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Veo 3.1 scores 1088 with a 95% confidence interval of plus or minus 7 over 7,529 blind pairwise votes. Placed by where that sits between Seedance 1.5 pro at 1000 and Gemini Omni Flash at 1233, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/video/leaderboard/text-to-video"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1720,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"video_generation","rating":1720,"weight":1}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"video_generation":{"elo":1720,"basisLine":"Estimated · Video generation · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":1720,"metrics":{"context_window":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/veo-3.1","verificationStatus":"verified","observedAt":"2026-09-21"},"max_output_tokens":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/veo-3.1","verificationStatus":"verified","observedAt":"2026-09-21"},"video_gen_time_sec":{"value":42.62,"unit":"s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/google/veo-3.1","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"qwen-audio-3-0-tts-plus","name":"Qwen-Audio-3.0-TTS Plus","provider":"Alibaba","providerId":"alibaba","presentation":{"task":"speech_synthesis","taskLabel":"Speech synthesis","modality":"audio","capabilities":{"inputs":["text"],"outputs":["speech"],"hasVision":false,"openWeights":false},"headline":{"value":1840,"label":"TTS Estimated rating","shortLabel":"TTS rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"speech_synthesis","isOverall":false},"categories":[{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"rated","sourceLabel":"Estimated","elo":1840,"coveragePct":0}],"facts":[],"speedClass":null,"price":null,"specifications":{"count":6,"published":4,"detailUrl":"/models/qwen-audio-3-0-tts-plus#specifications"},"detailUrl":"/models/qwen-audio-3-0-tts-plus"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"rated","score":84,"rating":1840,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"qwen-audio-3-0-tts-plus","categoryId":"speech_synthesis","score":84,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Qwen-Audio-3.0-TTS-Plus scores 1260 with a 95% confidence interval of plus or minus 16 over 1,559 blind pairwise votes. Placed by where that sits between Speech 2.8 HD at 1166 and Sonic 3.6 at 1276, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/text-to-speech/leaderboard/provider-voice"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":null},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"speech_synthesis":{"elo":1840,"basisLine":"Estimated · Speech synthesis · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":null,"metrics":{"max_output_tokens":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen-audio-3.0-tts-plus","verificationStatus":"verified","observedAt":"2026-09-21"},"input_price_per_m":{"value":20,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen-audio-3.0-tts-plus","verificationStatus":"verified","observedAt":"2026-09-21"},"context_window":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/qwen/qwen-audio-3.0-tts-plus","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"speech-2-8-hd","name":"Speech 2.8 HD","provider":"MiniMax","providerId":"minimax","presentation":{"task":"speech_synthesis","taskLabel":"Speech synthesis","modality":"audio","capabilities":{"inputs":["text"],"outputs":["speech"],"hasVision":false,"openWeights":false},"headline":{"value":1620,"label":"TTS Estimated rating","shortLabel":"TTS rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"speech_synthesis","isOverall":false},"categories":[{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0}],"facts":[],"speedClass":null,"price":null,"specifications":{"count":6,"published":4,"detailUrl":"/models/speech-2-8-hd#specifications"},"detailUrl":"/models/speech-2-8-hd"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"reasoning","label":"Reasoning","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"writing","label":"Writing","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"agents","label":"Agents & tool use","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"chat","label":"Chat & support","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"vision_understanding","label":"Vision understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":[],"excluded":[],"estimate":{"modelId":"speech-2-8-hd","categoryId":"speech_synthesis","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis arena for this modality, where Speech 2.8 HD scores 1166. Placed by where that sits between Speech 2.8 HD at 1166 and Sonic 3.6 at 1276, the span the board published on 2026-09-21, mapped onto 62-88. Arena Elo is blind human preference, not this category's recipe, so this is an ordering anchor rather than a measurement, and it is never compared against another modality's board. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/text-to-speech/leaderboard/provider-voice"],"confidence":"low","asOf":"2026-09-21","reviewOn":"2026-12-21"},"sourceLabel":"Estimated"},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":null},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":false,"publicationStatus":"public","seoStatus":"noindex","ratings":{"speech_synthesis":{"elo":1620,"basisLine":"Estimated · Speech synthesis · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-21","coveragePct":0,"opinion":true}},"speed":{},"overallElo":null,"metrics":{"context_window":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/speech-2.8-hd","verificationStatus":"verified","observedAt":"2026-09-21"},"input_price_per_m":{"value":100,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/speech-2.8-hd","verificationStatus":"verified","observedAt":"2026-09-21"},"max_output_tokens":{"value":0,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/minimax/speech-2.8-hd","verificationStatus":"verified","observedAt":"2026-09-21"}}},{"id":"command-a-plus","name":"Command A+","provider":"Cohere","providerId":"cohere","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1490,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1490,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1590,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1550,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":192000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":40,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":299,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.3,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":1.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":1.5,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/command-a-plus#specifications"},"detailUrl":"/models/command-a-plus"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":49,"rating":1490,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"command-a-plus","categoryId":"coding","score":49,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Command A+ scores 13, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/command-a-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"command-a-plus","categoryId":"reasoning","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Command A+ scores 13, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/command-a-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"command-a-plus","categoryId":"writing","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Command A+ scores 13, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/command-a-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":59,"rating":1590,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"command-a-plus","categoryId":"agents","score":59,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Command A+ scores 13, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/command-a-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"command-a-plus","categoryId":"chat","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Command A+ scores 13, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/command-a-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":55,"rating":1550,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"command-a-plus","categoryId":"vision_understanding","score":55,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Command A+ scores 13, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/command-a-plus"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1545,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1490,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1550,"weight":0.22727272727272727},{"categoryId":"writing","rating":1550,"weight":0.13636363636363635},{"categoryId":"agents","rating":1590,"weight":0.13636363636363635},{"categoryId":"chat","rating":1550,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1550,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1490,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"200–500ms"},"overallElo":1545,"metrics":{"context_window":{"value":192000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"},"output_price_per_m":{"value":1.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"},"cached_input_price_per_m":{"value":0.15,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"},"tokens_per_sec":{"value":40,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"},"input_price_per_m":{"value":0.3,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"},"ttft_ms":{"value":299,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"},"max_output_tokens":{"value":64000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/cohere/command-a-plus","verificationStatus":"verified","observedAt":"2026-09-23"}}},{"id":"gpt-6-luna","name":"GPT-6 Luna","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["file","image","text"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1580,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1580,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1630,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":62,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2294,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.1,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.5,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":0.5,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gpt-6-luna#specifications"},"detailUrl":"/models/gpt-6-luna"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":58,"rating":1580,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gpt-6-luna","categoryId":"coding","score":58,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Luna scores 37, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gpt-6-luna","categoryId":"reasoning","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Luna scores 37, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-6-luna","categoryId":"writing","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Luna scores 37, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":63,"rating":1630,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gpt-6-luna","categoryId":"agents","score":63,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Luna scores 37, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-6-luna","categoryId":"chat","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Luna scores 37, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gpt-6-luna","categoryId":"vision_understanding","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Luna scores 37, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-luna"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1627,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1580,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1640,"weight":0.22727272727272727},{"categoryId":"writing","rating":1640,"weight":0.13636363636363635},{"categoryId":"agents","rating":1630,"weight":0.13636363636363635},{"categoryId":"chat","rating":1640,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1640,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1580,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1627,"metrics":{"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"},"input_price_per_m":{"value":0.1,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"},"output_price_per_m":{"value":0.5,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"},"cached_input_price_per_m":{"value":0.01,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"},"ttft_ms":{"value":2294,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"},"tokens_per_sec":{"value":62,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-luna","verificationStatus":"verified","observedAt":"2026-09-23"}}},{"id":"gpt-6-sol","name":"GPT-6 Sol","provider":"OpenAI","providerId":"openai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["file","image","text"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1620,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1620,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1650,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1050000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":48,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":2918,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":2,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":10,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":10,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/gpt-6-sol#specifications"},"detailUrl":"/models/gpt-6-sol"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":62,"rating":1620,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"gpt-6-sol","categoryId":"coding","score":62,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Sol scores 48, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"gpt-6-sol","categoryId":"reasoning","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Sol scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-6-sol","categoryId":"writing","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Sol scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":65,"rating":1650,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"gpt-6-sol","categoryId":"agents","score":65,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Sol scores 48, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"gpt-6-sol","categoryId":"chat","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Sol scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"gpt-6-sol","categoryId":"vision_understanding","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which GPT-6 Sol scores 48, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/gpt-6-sol"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1664,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1620,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1680,"weight":0.22727272727272727},{"categoryId":"writing","rating":1680,"weight":0.13636363636363635},{"categoryId":"agents","rating":1650,"weight":0.13636363636363635},{"categoryId":"chat","rating":1680,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1680,"weight":0.09090909090909091}]}},"isFrontier":false,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1620,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1664,"metrics":{"output_price_per_m":{"value":10,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"},"ttft_ms":{"value":2918,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"},"input_price_per_m":{"value":2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"},"context_window":{"value":1050000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"},"cached_input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"},"tokens_per_sec":{"value":48,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/openai/gpt-6-sol","verificationStatus":"verified","observedAt":"2026-09-23"}}},{"id":"claude-opus-5-5","name":"Claude Opus 5.5","provider":"Anthropic","providerId":"anthropic","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1660,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1720,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1720,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1660,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1720,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1720,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1000000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":72,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":4222,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":4,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":20,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":20,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/claude-opus-5-5#specifications"},"detailUrl":"/models/claude-opus-5-5"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"claude-opus-5-5","categoryId":"coding","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5.5 scores 58, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":72,"rating":1720,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"claude-opus-5-5","categoryId":"reasoning","score":72,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5.5 scores 58, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":72,"rating":1720,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-opus-5-5","categoryId":"writing","score":72,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5.5 scores 58, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":66,"rating":1660,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"claude-opus-5-5","categoryId":"agents","score":66,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5.5 scores 58, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":72,"rating":1720,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"claude-opus-5-5","categoryId":"chat","score":72,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5.5 scores 58, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":72,"rating":1720,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"claude-opus-5-5","categoryId":"vision_understanding","score":72,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Claude Opus 5.5 scores 58, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/claude-opus-5-5"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1700,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1660,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1720,"weight":0.22727272727272727},{"categoryId":"writing","rating":1720,"weight":0.13636363636363635},{"categoryId":"agents","rating":1660,"weight":0.13636363636363635},{"categoryId":"chat","rating":1720,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1720,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1660,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"2s+"},"overallElo":1700,"metrics":{"tokens_per_sec":{"value":72,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"},"cached_input_price_per_m":{"value":0.2,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"},"max_output_tokens":{"value":128000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"},"output_price_per_m":{"value":20,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"},"context_window":{"value":1000000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"},"input_price_per_m":{"value":4,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"},"ttft_ms":{"value":4222,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/anthropic/claude-opus-5.5","verificationStatus":"verified","observedAt":"2026-09-23"}}},{"id":"mimo-v2-6-pro","name":"MiMo-V2.6-Pro","provider":"Xiaomi","providerId":"xiaomi","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","video","audio"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1610,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":1048576,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":24,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":5839,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":0.435,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":0.87,"unit":"$/1M tok","kind":"cost"}],"speedClass":"<50 tok/s","price":{"value":0.87,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/mimo-v2-6-pro#specifications"},"detailUrl":"/models/mimo-v2-6-pro"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"mimo-v2-6-pro","categoryId":"coding","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiMo-V2.6-Pro scores 46, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mimo-v2-6-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"mimo-v2-6-pro","categoryId":"reasoning","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiMo-V2.6-Pro scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mimo-v2-6-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"mimo-v2-6-pro","categoryId":"writing","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiMo-V2.6-Pro scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mimo-v2-6-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"mimo-v2-6-pro","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiMo-V2.6-Pro scores 46, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mimo-v2-6-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"mimo-v2-6-pro","categoryId":"chat","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiMo-V2.6-Pro scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mimo-v2-6-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"mimo-v2-6-pro","categoryId":"vision_understanding","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which MiMo-V2.6-Pro scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/mimo-v2-6-pro"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1661,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1610,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1680,"weight":0.22727272727272727},{"categoryId":"writing","rating":1680,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1680,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1680,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1610,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"<50 tok/s","latencyClass":"2s+"},"overallElo":1661,"metrics":{"context_window":{"value":1048576,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"},"ttft_ms":{"value":5839,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"},"output_price_per_m":{"value":0.87,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"},"tokens_per_sec":{"value":24,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"},"input_price_per_m":{"value":0.435,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"},"max_output_tokens":{"value":131072,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"},"cached_input_price_per_m":{"value":0.0036,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/xiaomi/mimo-v2.6-pro","verificationStatus":"verified","observedAt":"2026-09-23"}}},{"id":"grok-4-7","name":"Grok 4.7","provider":"xAI","providerId":"xai","presentation":{"task":"vision","taskLabel":"Vision-language","modality":"text","capabilities":{"inputs":["text","image","file"],"outputs":["text"],"hasVision":true,"openWeights":false},"headline":{"value":1610,"label":"Coding Estimated rating","shortLabel":"Coding rating","basis":{"text":"Estimated rating · thin coverage (0% of recipe inputs)","caution":true},"categoryId":"coding","isOverall":false},"categories":[{"categoryId":"coding","label":"Coding","state":"rated","sourceLabel":"Estimated","elo":1610,"coveragePct":0},{"categoryId":"reasoning","label":"Reasoning","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"writing","label":"Writing","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"agents","label":"Agents & tool use","state":"rated","sourceLabel":"Estimated","elo":1640,"coveragePct":0},{"categoryId":"chat","label":"Chat & support","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","sourceLabel":"Estimated","elo":1680,"coveragePct":0}],"facts":[{"metric":"context_window","label":"Context (tokens)","value":500000,"unit":"tokens","kind":"capacity"},{"metric":"tokens_per_sec","label":"Speed (tok/s)","value":61,"unit":"tok/s","kind":"speed"},{"metric":"ttft_ms","label":"TTFT","value":1288,"unit":"ms","kind":"speed"},{"metric":"input_price_per_m","label":"In $/1M tok","value":1.6,"unit":"$/1M tok","kind":"cost"},{"metric":"output_price_per_m","label":"Out $/1M tok","value":4.8,"unit":"$/1M tok","kind":"cost"}],"speedClass":"50–100 tok/s","price":{"value":4.8,"unit":"USD/1M output tokens"},"specifications":{"count":10,"published":9,"detailUrl":"/models/grok-4-7#specifications"},"detailUrl":"/models/grok-4-7"},"reportedRatings":{"basisKind":"reported","configurationScope":"reported_configurations","ratingVersion":"reported-with-estimates-v2","releaseVersion":"recent-models-2026-09-23-r1","recipeVersion":"reported-text-2026-09-11-r1","researchedThrough":"2026-09-23","label":"Estimated ratings","caveat":"Models may use different benchmarks and test settings. This is an indicative composite, not a controlled head-to-head comparison or community Elo. Admin-approved sentiment estimates fill categories without accepted benchmark results. Estimates are labelled and do not increase benchmark coverage.","assessments":[{"categoryId":"coding","label":"Coding","state":"rated","score":61,"rating":1610,"coveragePct":0,"inputs":[],"missingFamilies":["repository-engineering","terminal-work","frontiercode"],"excluded":[],"estimate":{"modelId":"grok-4-7","categoryId":"coding","score":61,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.7 scores 46, and placed within the 48-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-7"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"reasoning","label":"Reasoning","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["graduate-science","advanced-mathematics","broad-academic","interactive-abstraction"],"excluded":[],"estimate":{"modelId":"grok-4-7","categoryId":"reasoning","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.7 scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-7"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"writing","label":"Writing","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["constrained-story-writing","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-7","categoryId":"writing","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.7 scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-7"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"agents","label":"Agents & tool use","state":"rated","score":64,"rating":1640,"coveragePct":0,"inputs":[],"missingFamilies":["professional-computer-tasks","desktop-use","workflow-automation","web-research"],"excluded":[],"estimate":{"modelId":"grok-4-7","categoryId":"agents","score":64,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.7 scores 46, and placed within the 59-66 band this catalog's measured results occupy for this category. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-7"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"chat","label":"Chat & support","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["conversation-quality","support-task-completion","instruction-compliance"],"excluded":[],"estimate":{"modelId":"grok-4-7","categoryId":"chat","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.7 scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-7"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"vision_understanding","label":"Vision understanding","state":"rated","score":68,"rating":1680,"coveragePct":0,"inputs":[],"missingFamilies":["visual-grounding","spatial-reconstruction","visual-symbol-recognition"],"excluded":[],"estimate":{"modelId":"grok-4-7","categoryId":"vision_understanding","score":68,"rationale":"Editorial estimate, anchored to the Artificial Analysis Intelligence Index v4.3, on which Grok 4.7 scores 46, and placed mid-range, because this category has too few measured results to define a band. AA's index is a composite of ten evaluations and is not this category's recipe, so this is an ordering anchor rather than a measurement. Any accepted result that clears the evidence thresholds replaces it.","sourceUrls":["https://artificialanalysis.ai/models/grok-4-7"],"confidence":"low","asOf":"2026-09-23","reviewOn":"2026-12-23"},"sourceLabel":"Estimated"},{"categoryId":"image_generation","label":"Image generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_generation","label":"Video generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"video_understanding","label":"Video understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"audio_understanding","label":"Audio understanding","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_recognition","label":"Speech recognition","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"speech_synthesis","label":"Speech synthesis","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]},{"categoryId":"music_generation","label":"Music generation","state":"not_applicable","score":null,"rating":null,"coveragePct":null,"inputs":[],"missingFamilies":[],"excluded":[]}],"overall":{"rating":1661,"coveragePct":0,"sourceLabel":"Estimated","contributors":[{"categoryId":"coding","rating":1610,"weight":0.22727272727272727},{"categoryId":"reasoning","rating":1680,"weight":0.22727272727272727},{"categoryId":"writing","rating":1680,"weight":0.13636363636363635},{"categoryId":"agents","rating":1640,"weight":0.13636363636363635},{"categoryId":"chat","rating":1680,"weight":0.18181818181818182},{"categoryId":"vision_understanding","rating":1680,"weight":0.09090909090909091}]}},"isFrontier":true,"isOpenSource":false,"status":"active","indexable":true,"publicationStatus":"public","seoStatus":"eligible","ratings":{"coding":{"elo":1610,"basisLine":"Estimated · Coding · 0 benchmark families · 0% benchmark coverage · low confidence · review 2026-12-23","coveragePct":0,"opinion":true}},"speed":{"throughputClass":"50–100 tok/s","latencyClass":"500ms–2s"},"overallElo":1661,"metrics":{"max_output_tokens":{"value":450000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"},"input_price_per_m":{"value":1.6,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"},"cached_input_price_per_m":{"value":0.4,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"},"tokens_per_sec":{"value":61,"unit":"tok/s","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"},"ttft_ms":{"value":1288,"unit":"ms","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"},"output_price_per_m":{"value":4.8,"unit":"$/1M tok","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"},"context_window":{"value":500000,"unit":"tokens","source":"Source not recorded","evidenceTier":"legacy","sourceUrl":"https://openrouter.ai/x-ai/grok-4.7","verificationStatus":"verified","observedAt":"2026-09-23"}}}]}