{
  "meta": {
    "api_version": "v1",
    "endpoint": "/api/v1/recommendations",
    "updated": "2026-09-12",
    "ranking_kind": "benchr editorial heuristic — decision support, not an official benchmark or a claim of the objectively best model",
    "assumptions": {
      "task": "chat",
      "budget": "any",
      "priority": "balanced",
      "privacy": "no",
      "provider": null,
      "blended_price": "70% input + 30% output price per million tokens; does not include cache, batch, taxes, tool fees, or self-hosting infrastructure.",
      "quality": "benchr editorial capability ratings.",
      "speed": "benchr editorial first-token and throughput estimates.",
      "privacy_note": "No license filter."
    },
    "candidates_after_filters": 38,
    "returned": 5
  },
  "data": [
    {
      "rank": 1,
      "model": {
        "id": "qwen-3-6-27b",
        "name": "Qwen3.6-27B",
        "provider": "Alibaba (Qwen)",
        "api_name": "Qwen/Qwen3.6-27B",
        "license": "Apache-2.0",
        "type": "open"
      },
      "editorial_score": 84.22,
      "score_components": {
        "task_quality": 87.7,
        "affordability": 100,
        "speed": 40,
        "openness": 100
      },
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Multilingual coding (Chinese, Japanese, Korean, Arabic)",
        "Local inference on consumer GPUs — dense 27B",
        "Tool-use agent loops at zero API cost"
      ],
      "skip_if": [
        "You want a managed hosted API — these are open weights you self-host",
        "Absolute deepest single-language reasoning"
      ],
      "official_evidence": {
        "record_id": "qwen-3-6-27b",
        "api_model_id": "Qwen/Qwen3.6-27B",
        "verified_date": "2026-09-11",
        "sources": {
          "release": "https://huggingface.co/Qwen/Qwen3.6-27B",
          "license": "https://huggingface.co/Qwen/Qwen3.6-27B",
          "context": "https://huggingface.co/Qwen/Qwen3.6-27B",
          "benchmarks": "https://huggingface.co/Qwen/Qwen3.6-27B"
        },
        "pricing": {
          "selfHost": true,
          "inputPerM": null,
          "outputPerM": null
        },
        "context": {
          "windowTokens": 262144,
          "windowTokensExtended": 1010000,
          "maxOutputTokens": null
        },
        "benchmarks": {
          "SWE-bench Verified": 77.2,
          "SWE-bench Pro": 53.5,
          "Terminal-Bench 2.0": 59.3,
          "MMLU-Pro": 86.2,
          "GPQA Diamond": 87.8,
          "AIME 2026": 94.1,
          "MMMU": 82.9
        },
        "notes": "Open weight, Apache-2.0, self-host (no per-token list price). 262,144 native context, extensible ~1M via YaRN. Benchmarks from the official HF model card."
      }
    },
    {
      "rank": 2,
      "model": {
        "id": "llama-4-maverick",
        "name": "Llama 4 Maverick",
        "provider": "Meta",
        "api_name": "meta-llama/Llama-4-Maverick-17B-128E-Instruct",
        "license": "Llama 4 Community License",
        "type": "frontier-open"
      },
      "editorial_score": 79.45,
      "score_components": {
        "task_quality": 79,
        "affordability": 100,
        "speed": 40,
        "openness": 100
      },
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Self-hosting under the Llama 4 Community License with a 1M-token window",
        "Self-hosted production at zero licensing cost",
        "Multimodal at no API cost"
      ],
      "skip_if": [
        "You need the very best reasoning or coding",
        "You can't manage GPU infrastructure"
      ],
      "official_evidence": {
        "record_id": "llama-4-maverick",
        "api_model_id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct",
        "verified_date": "2026-09-11",
        "sources": {
          "release": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
          "license": "https://www.llama.com/llama4/license/",
          "context": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct",
          "benchmarks": "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct"
        },
        "pricing": {
          "selfHost": true,
          "inputPerM": null,
          "outputPerM": null
        },
        "context": {
          "windowTokens": 1000000,
          "maxOutputTokens": null
        },
        "benchmarks": {
          "MMLU-Pro (0-shot)": 80.5,
          "GPQA Diamond": 69.8,
          "LiveCodeBench": 43.4,
          "MGSM": 92.3
        },
        "notes": "Open weight; $0 to self-host. 1M-token context, 128 experts. Llama 4 Behemoth (288B active / ~2T total) was only ever previewed as 'still training' and was never released — do not list specs for it as a usable model."
      }
    },
    {
      "rank": 3,
      "model": {
        "id": "deepseek-v4-pro",
        "name": "DeepSeek V4-Pro",
        "provider": "DeepSeek",
        "api_name": "deepseek-v4-pro",
        "license": "MIT",
        "type": "frontier-open"
      },
      "editorial_score": 78.3,
      "score_components": {
        "task_quality": 87.7,
        "affordability": 76.3,
        "speed": 40,
        "openness": 100
      },
      "pricing": {
        "input_per_million": 1.32,
        "output_per_million": 3.96,
        "cache_input_per_million": 0.044,
        "batch_discount": null,
        "self_hosted": true,
        "offpeak_input_per_million": 0.66,
        "offpeak_output_per_million": 1.98,
        "offpeak_cache_input_per_million": 0.022,
        "pricing_note": "Peak (list) rate. DeepSeek bills 01:00-04:00 and 06:00-10:00 UTC Monday to Friday at this rate and every other hour at half of it, effective 16:00 UTC on 2026-08-16."
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Frontier-grade open-weight coding",
        "Math-heavy work",
        "Self-hosted production, where the August 2026 API price rise does not apply"
      ],
      "skip_if": [
        "You need vision/multimodal",
        "You can't manage GPU hosting"
      ],
      "official_evidence": {
        "record_id": "deepseek-v4-pro",
        "api_model_id": "deepseek-v4-pro",
        "verified_date": "2026-09-11",
        "sources": {
          "release": "https://api-docs.deepseek.com/news/news260424",
          "pricing": "https://api-docs.deepseek.com/quick_start/pricing",
          "license": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
          "context": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
          "benchmarks": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
          "ga_update": "https://api-docs.deepseek.com/updates",
          "lifecycle": "https://api-docs.deepseek.com/news/news"
        },
        "pricing": {
          "inputPerM": 1.32,
          "outputPerM": 3.96,
          "cacheHitInputPerM": 0.044,
          "offPeakInputPerM": 0.66,
          "offPeakOutputPerM": 1.98,
          "offPeakCacheHitInputPerM": 0.022,
          "offPeakDiscount": 0.5,
          "peakHoursUTC": "01:00-04:00 and 06:00-10:00, Monday to Friday",
          "priceEffectiveFrom": "2026-08-16T16:00:00Z",
          "previousFlatPricing": {
            "inputPerM": 0.435,
            "outputPerM": 0.87,
            "cacheHitInputPerM": 0.003625
          },
          "selfHost": true,
          "note": "DeepSeek moved to peak/off-peak billing at 16:00 UTC on 2026-08-16. Peak hours are 01:00-04:00 and 06:00-10:00 UTC, Monday to Friday; every other hour bills at half the peak rate. The figures in inputPerM/outputPerM/cacheHitInputPerM are the peak (list) rates; the offPeak* fields are the discounted rates. Read on api-docs.deepseek.com/quick_start/pricing on 2026-08-28."
        },
        "context": {
          "windowTokens": 1000000,
          "maxOutputTokens": 384000
        },
        "benchmarks": {
          "SWE-bench Verified": 80.6,
          "SWE-bench Pro": 55.4,
          "GPQA Diamond": 90.1,
          "LiveCodeBench": 93.5,
          "Terminal-Bench 2.0": 67.9,
          "MMLU-Pro (Max)": 87.5,
          "MRCR 1M": 83.5,
          "Codeforces (rating)": 3206,
          "Terminal-Bench 2.1": 87.9,
          "Humanity's Last Exam (no tools)": 42.7,
          "Humanity's Last Exam (with tools)": 60,
          "NL2Repo": 61.5,
          "CyberGym": 83.3,
          "DeepSWE": 62.7,
          "Toolathlon-Verified": 74.1,
          "DSBench-Hard": 67.2
        },
        "notes": "UPDATED 2026-08-28: DeepSeek raised API prices and introduced peak/off-peak billing effective 16:00 UTC on August 16, 2026. Cache-miss input went from a flat $0.435 to $1.32 peak / $0.66 off-peak per 1M tokens, output from $0.87 to $3.96 / $1.98, and cache-hit input from $0.003625 to $0.044 / $0.022 - the steepest line on the sheet at roughly twelve times the old cache-hit rate. Peak hours are 01:00-04:00 and 06:00-10:00 UTC, Monday to Friday. DeepSeek also shipped a V4-Pro GA update on August 13, 2026 with a new provider-reported benchmark table (Terminal Bench 2.1 87.9, HLE 42.7 without tools / 60.0 with tools, NL2Repo 61.5, CyberGym 83.3, DeepSWE 62.7, Toolathlon-Verified 74.1, DSBench-Hard 67.2); those are DeepSeek's own numbers, not benchr tests. The April model-card figures are retained as published. Open weights, MIT. RECHECKED 2026-09-11: the pricing page lists deepseek-v4-pro as version DeepSeek-V4-Pro-0813 at the same $1.32 / $3.96 peak and $0.66 / $1.98 off-peak rates, 1M context and 384K output. DeepSeek's September 10, 2026 release note announced that deepseek-v4-pro requests would route to V4.1-Flash from 04:00 UTC on September 14; a later entry on DeepSeek's news page withdrew that, saying it will \"continue providing API services for DeepSeek V4 Pro after September 14, 2026, with the billing method remaining unchanged.\" The withdrawn date is recorded here rather than as a deprecation."
      }
    },
    {
      "rank": 4,
      "model": {
        "id": "mistral-large-3",
        "name": "Mistral Large 3",
        "provider": "Mistral AI",
        "api_name": "mistral-large-2512",
        "license": "Apache-2.0",
        "type": "frontier-open"
      },
      "editorial_score": 77.85,
      "score_components": {
        "task_quality": 81.7,
        "affordability": 87.7,
        "speed": 40,
        "openness": 100
      },
      "pricing": {
        "input_per_million": 0.5,
        "output_per_million": 1.5,
        "cache_input_per_million": null,
        "batch_discount": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Apache-licensed production workloads",
        "European data residency",
        "Very cheap inference with decent reasoning"
      ],
      "skip_if": [
        "Coding at frontier quality",
        "Vision or multimodal workflows"
      ],
      "official_evidence": {
        "record_id": "mistral-large-3",
        "api_model_id": "mistral-large-2512",
        "verified_date": "2026-06-12",
        "sources": {
          "release": "https://docs.mistral.ai/models/mistral-large-3-25-12",
          "pricing": "https://docs.mistral.ai/models/mistral-large-3-25-12",
          "license": "https://mistral.ai/news/mistral-3/",
          "context": "https://docs.mistral.ai/models/mistral-large-3-25-12",
          "benchmarks": "https://mistral.ai/news/mistral-3/"
        },
        "pricing": {
          "inputPerM": 0.5,
          "outputPerM": 1.5,
          "cacheHitInputPerM": null,
          "selfHost": true
        },
        "context": {
          "windowTokens": 256000,
          "maxOutputTokens": null
        },
        "benchmarks": {},
        "notes": "Open weights, Apache-2.0. Mistral's announcement gives only relative/leaderboard claims for Large 3, no discrete official per-benchmark scores (the ~85% AIME figure on the page belongs to a smaller reasoning variant, NOT Large 3) — so benchmarks is intentionally empty, not guessed. Cache-hit price not published."
      }
    },
    {
      "rank": 5,
      "model": {
        "id": "phi-4",
        "name": "Phi-4",
        "provider": "Microsoft",
        "api_name": "microsoft/phi-4",
        "license": "MIT",
        "type": "small-open"
      },
      "editorial_score": 77.07,
      "score_components": {
        "task_quality": 74.7,
        "affordability": 100,
        "speed": 40,
        "openness": 100
      },
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": null,
        "tokens_per_second": null,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Local inference on consumer hardware",
        "Edge deployment",
        "Reasoning at tiny scale"
      ],
      "skip_if": [
        "Long-context tasks",
        "Production-quality writing or coding"
      ],
      "official_evidence": {
        "record_id": "phi-4",
        "api_model_id": "microsoft/phi-4",
        "verified_date": "2026-09-11",
        "sources": {
          "release": "https://huggingface.co/microsoft/phi-4",
          "license": "https://huggingface.co/microsoft/phi-4",
          "context": "https://huggingface.co/microsoft/phi-4",
          "benchmarks": "https://www.microsoft.com/en-us/research/publication/phi-4-technical-report/"
        },
        "pricing": {
          "selfHost": true,
          "inputPerM": null,
          "outputPerM": null
        },
        "context": {
          "windowTokens": 16000,
          "maxOutputTokens": null
        },
        "benchmarks": {
          "GPQA Diamond": 56.1,
          "MMLU": 84.8,
          "HumanEval": 82.6,
          "MATH": 80.4
        },
        "notes": "Verified 2026-06-03: MIT-licensed, 14B dense, 16K context, self-host only (no Microsoft per-token API; available on Azure AI Foundry + Hugging Face). Released 2024-12-12. Benchmarks from the Phi-4 technical report / model card."
      }
    }
  ]
}