{
  "meta": {
    "api_version": "v1",
    "endpoint": "/api/v1/recommendations",
    "updated": "2026-07-29",
    "ranking_kind": "benchr editorial heuristic — decision support, not an official benchmark or a claim of the objectively best model",
    "assumptions": {
      "task": "chat",
      "budget": "any",
      "priority": "balanced",
      "privacy": "no",
      "provider": null,
      "blended_price": "70% input + 30% output price per million tokens; does not include cache, batch, taxes, tool fees, or self-hosting infrastructure.",
      "quality": "benchr editorial capability ratings.",
      "speed": "benchr editorial first-token and throughput estimates.",
      "privacy_note": "No license filter."
    },
    "candidates_after_filters": 29,
    "returned": 5
  },
  "data": [
    {
      "rank": 1,
      "model": {
        "id": "qwen-3-6-27b",
        "name": "Qwen3.6-27B",
        "provider": "Alibaba",
        "api_name": "Qwen/Qwen3.6-27B",
        "license": "Apache 2.0",
        "type": "open"
      },
      "editorial_score": 87.84,
      "score_components": {
        "task_quality": 87.7,
        "affordability": 100,
        "speed": 64.2,
        "openness": 100
      },
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": 340,
        "tokens_per_second": 105,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Multilingual coding (Chinese, Japanese, Korean, Arabic)",
        "Local inference on consumer GPUs — dense 27B",
        "Tool-use agent loops at zero API cost"
      ],
      "skip_if": [
        "You want a managed hosted API — these are open weights you self-host",
        "Absolute deepest single-language reasoning"
      ],
      "official_evidence": {
        "record_id": "qwen-3-6-27b",
        "api_model_id": "Qwen/Qwen3.6-27B",
        "verified_date": "2026-05-31",
        "sources": {
          "release": "https://huggingface.co/Qwen/Qwen3.6-27B",
          "license": "https://huggingface.co/Qwen/Qwen3.6-27B",
          "benchmarks": "https://huggingface.co/Qwen/Qwen3.6-27B"
        },
        "pricing": {
          "selfHost": true,
          "inputPerM": null,
          "outputPerM": null
        },
        "context": {
          "windowTokens": 262144,
          "windowTokensExtended": 1010000,
          "maxOutputTokens": null
        },
        "benchmarks": {
          "SWE-bench Verified": 77.2,
          "SWE-bench Pro": 53.5,
          "Terminal-Bench 2.0": 59.3,
          "MMLU-Pro": 86.2,
          "GPQA Diamond": 87.8,
          "AIME 2026": 94.1,
          "MMMU": 82.9
        },
        "notes": "Open weight, Apache-2.0, self-host (no per-token list price). 262,144 native context, extensible ~1M via YaRN. Benchmarks from the official HF model card."
      }
    },
    {
      "rank": 2,
      "model": {
        "id": "deepseek-v4-flash",
        "name": "DeepSeek V4-Flash",
        "provider": "DeepSeek",
        "api_name": "deepseek-v4-flash",
        "license": "MIT",
        "type": "open"
      },
      "editorial_score": 86.55,
      "score_components": {
        "task_quality": 84,
        "affordability": 96.5,
        "speed": 74.8,
        "openness": 100
      },
      "pricing": {
        "input_per_million": 0.14,
        "output_per_million": 0.28,
        "cache_input_per_million": 0.0028,
        "batch_discount": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": 290,
        "tokens_per_second": 135,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Lowest listed hosted-model inference rate among models tracked by benchr in the July 24, 2026 snapshot; re-check live provider pricing before budgeting",
        "Self-hosted at low cost",
        "Volume coding tasks"
      ],
      "skip_if": [
        "You need frontier-tier reasoning depth",
        "Vision-first workflows"
      ],
      "official_evidence": {
        "record_id": "deepseek-v4-flash",
        "api_model_id": "deepseek-v4-flash",
        "verified_date": "2026-07-24",
        "sources": {
          "release": "https://api-docs.deepseek.com/news/news260424",
          "pricing": "https://api-docs.deepseek.com/quick_start/pricing",
          "license": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash",
          "benchmarks": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash"
        },
        "pricing": {
          "inputPerM": 0.14,
          "outputPerM": 0.28,
          "cacheHitInputPerM": 0.0028,
          "selfHost": true
        },
        "context": {
          "windowTokens": 1000000,
          "maxOutputTokens": 384000
        },
        "benchmarks": {
          "SWE-bench Verified": 79,
          "GPQA Diamond": 88.1,
          "LiveCodeBench": 91.6,
          "MMLU-Pro (Think Max)": 86.2,
          "HMMT 2026 Feb": 94.8,
          "MRCR 1M": 78.7
        },
        "notes": "Official pricing checked 2026-07-24: $0.14 input / $0.28 output / $0.0028 cache-hit input per 1M tokens. Open weights, MIT. Benchmarks are DeepSeek's provider-reported model-card figures (Instruct / Think Max mode), not independently reproduced results."
      }
    },
    {
      "rank": 3,
      "model": {
        "id": "phi-4",
        "name": "Phi-4",
        "provider": "Microsoft",
        "api_name": "phi-4",
        "license": "MIT",
        "type": "small-open"
      },
      "editorial_score": 86.07,
      "score_components": {
        "task_quality": 74.7,
        "affordability": 100,
        "speed": 100,
        "openness": 100
      },
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": 95,
        "tokens_per_second": 220,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Local inference on consumer hardware",
        "Edge deployment",
        "Reasoning at tiny scale"
      ],
      "skip_if": [
        "Long-context tasks",
        "Production-quality writing or coding"
      ],
      "official_evidence": {
        "record_id": "phi-4",
        "api_model_id": "microsoft/phi-4",
        "verified_date": "2026-06-03",
        "sources": {
          "release": "https://huggingface.co/microsoft/phi-4",
          "license": "https://huggingface.co/microsoft/phi-4",
          "context": "https://huggingface.co/microsoft/phi-4",
          "benchmarks": "https://www.microsoft.com/en-us/research/publication/phi-4-technical-report/"
        },
        "pricing": {
          "selfHost": true,
          "inputPerM": null,
          "outputPerM": null
        },
        "context": {
          "windowTokens": 16000,
          "maxOutputTokens": null
        },
        "benchmarks": {
          "GPQA Diamond": 56.1,
          "MMLU": 84.8,
          "HumanEval": 82.6,
          "MATH": 80.4
        },
        "notes": "Verified 2026-06-03: MIT-licensed, 14B dense, 16K context, self-host only (no Microsoft per-token API; available on Azure AI Foundry + Hugging Face). Released 2024-12-12. Benchmarks from the Phi-4 technical report / model card."
      }
    },
    {
      "rank": 4,
      "model": {
        "id": "deepseek-v4-pro",
        "name": "DeepSeek V4-Pro",
        "provider": "DeepSeek",
        "api_name": "deepseek-v4-pro",
        "license": "MIT",
        "type": "frontier-open"
      },
      "editorial_score": 84.86,
      "score_components": {
        "task_quality": 87.7,
        "affordability": 90.7,
        "speed": 59.8,
        "openness": 100
      },
      "pricing": {
        "input_per_million": 0.435,
        "output_per_million": 0.87,
        "cache_input_per_million": 0.003625,
        "batch_discount": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": 380,
        "tokens_per_second": 95,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Frontier-grade open-weight coding",
        "Math-heavy work",
        "Self-hosted production at near-zero API cost"
      ],
      "skip_if": [
        "You need vision/multimodal",
        "You can't manage GPU hosting"
      ],
      "official_evidence": {
        "record_id": "deepseek-v4-pro",
        "api_model_id": "deepseek-v4-pro",
        "verified_date": "2026-07-24",
        "sources": {
          "release": "https://api-docs.deepseek.com/news/news260424",
          "pricing": "https://api-docs.deepseek.com/quick_start/pricing",
          "license": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
          "benchmarks": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro"
        },
        "pricing": {
          "inputPerM": 0.435,
          "outputPerM": 0.87,
          "cacheHitInputPerM": 0.003625,
          "selfHost": true,
          "priceChangeNote": "The official pricing page still listed $0.435 input / $0.87 output / $0.003625 cache-hit input per 1M tokens when checked on 2026-07-24. DeepSeek states that product prices may vary and reserves the right to adjust them, so these current listed rates should not be treated as a guarantee of future pricing."
        },
        "context": {
          "windowTokens": 1000000,
          "maxOutputTokens": 384000
        },
        "benchmarks": {
          "SWE-bench Verified": 80.6,
          "SWE-bench Pro": 55.4,
          "GPQA Diamond": 90.1,
          "LiveCodeBench": 93.5,
          "Terminal-Bench 2.0": 67.9,
          "MMLU-Pro (Max)": 87.5,
          "MRCR 1M": 83.5,
          "Codeforces (rating)": 3206
        },
        "notes": "Official pricing checked 2026-07-24: $0.435 input / $0.87 output / $0.003625 cache-hit input per 1M tokens. The listed rate remained at the post-promotion level, but DeepSeek reserves the right to adjust product prices. SWE-bench Verified 80.6% is DeepSeek's provider-reported model-card figure, not an independently reproduced result. Open weights, MIT."
      }
    },
    {
      "rank": 5,
      "model": {
        "id": "llama-4-scout",
        "name": "Llama 4 Scout",
        "provider": "Meta",
        "api_name": "llama-4-scout",
        "license": "Llama 4 Community",
        "type": "open"
      },
      "editorial_score": 84.5,
      "score_components": {
        "task_quality": 74,
        "affordability": 100,
        "speed": 92,
        "openness": 100
      },
      "pricing": {
        "input_per_million": null,
        "output_per_million": null,
        "self_hosted": true
      },
      "latency": {
        "first_token_ms": 180,
        "tokens_per_second": 180,
        "provenance": "benchr editorial tool estimate; not a controlled lab measurement."
      },
      "best_for": [
        "Ultra-long-context tasks (10M token window)",
        "Fast self-hosted inference",
        "Free multimodal at scale"
      ],
      "skip_if": [
        "Deep reasoning tasks",
        "When frontier coding quality is needed"
      ],
      "official_evidence": {
        "record_id": "llama-4-scout",
        "api_model_id": "meta-llama/Llama-4-Scout-17B-16E",
        "verified_date": "2026-05-31",
        "sources": {
          "release": "https://ai.meta.com/blog/llama-4-multimodal-intelligence/",
          "license": "https://www.llama.com/llama4/license/",
          "context": "https://www.llama.com/docs/model-cards-and-prompt-formats/llama4/",
          "benchmarks": "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E"
        },
        "pricing": {
          "selfHost": true,
          "inputPerM": null,
          "outputPerM": null
        },
        "context": {
          "windowTokens": 10000000,
          "maxOutputTokens": null
        },
        "benchmarks": {
          "MMLU-Pro (0-shot)": 74.3,
          "GPQA Diamond": 57.2,
          "MMMU": 73.4,
          "MathVista": 73.7,
          "LiveCodeBench": 32.8
        },
        "notes": "Open weight; $0 to self-host (Meta sells no API). Community License: >700M MAU needs a separate Meta license; not OSI-approved. 10M-token context. Benchmarks from Meta's official instruction-tuned model card."
      }
    }
  ]
}