{
  "_meta": {
    "title": "benchr use-case evaluation packs",
    "version": "1.0.0",
    "updated": "2026-08-21",
    "provenance": "Original fixed test cases and scoring rubrics authored for benchr. They are evaluation materials, not model outputs, benchmark results, or claims that benchr ran the models.",
    "license": "CC BY 4.0",
    "attribution": "benchr.org",
    "privacy": "The file contains synthetic prompts and rubrics only. Benchr Labs processes user-entered outputs locally in the browser."
  },
  "suites": [
    {
      "id": "professional-writing-v1",
      "name": "Professional writing workflows",
      "name_ar": "حزمة الكتابة المهنية",
      "locale": "en-ar",
      "available_in": ["en", "ar"],
      "version": "1.0.0",
      "description": "Five fixed cases for long-form structure, email restraint, resume integrity, source preservation, and social repurposing.",
      "description_ar": "خمس حالات ثابتة للبنية الطويلة، وضبط البريد، ونزاهة السيرة، وحفظ المصادر، وإعادة توظيف المحتوى.",
      "rubric": [
        {
          "id": "instruction_following",
          "label": "Instruction following",
          "label_ar": "اتباع التعليمات",
          "description": "Follows every explicit content, length, and format constraint.",
          "description_ar": "يلتزم بقيود المحتوى والطول والتنسيق كلها.",
          "anchors": {"1": "Misses major constraints", "3": "Mostly follows them", "5": "Follows every constraint"},
          "anchors_ar": {"1": "يتجاهل قيودًا أساسية", "3": "يلتزم بمعظمها", "5": "يلتزم بها كلها"}
        },
        {
          "id": "factual_integrity",
          "label": "Factual integrity",
          "label_ar": "النزاهة الواقعية",
          "description": "Preserves supplied facts and does not invent achievements, dates, or evidence.",
          "description_ar": "يحفظ الحقائق المعطاة ولا يختلق إنجازات أو تواريخ أو أدلة.",
          "anchors": {"1": "Invents or changes facts", "3": "Minor unsupported wording", "5": "Every claim is supported"},
          "anchors_ar": {"1": "يختلق أو يغيّر الحقائق", "3": "صياغة بسيطة بلا سند", "5": "كل ادعاء مسند"}
        },
        {
          "id": "audience_fit",
          "label": "Audience fit",
          "label_ar": "ملاءمة الجمهور",
          "description": "Uses the requested register, purpose, and level of detail.",
          "description_ar": "يستخدم النبرة والغرض ومستوى التفصيل المطلوب.",
          "anchors": {"1": "Wrong audience", "3": "Usable with editing", "5": "Ready for the audience"},
          "anchors_ar": {"1": "لا يناسب الجمهور", "3": "يصلح بعد تحرير", "5": "جاهز للجمهور"}
        },
        {
          "id": "revision_value",
          "label": "Revision value",
          "label_ar": "قيمة التحرير",
          "description": "Improves clarity and structure without flattening the source voice.",
          "description_ar": "يحسن الوضوح والبنية من دون محو صوت النص الأصلي.",
          "anchors": {"1": "Worse than the source", "3": "Some useful edits", "5": "Materially better and faithful"},
          "anchors_ar": {"1": "أسوأ من الأصل", "3": "تعديلات مفيدة جزئيًا", "5": "أفضل بوضوح وأمين للأصل"}
        }
      ],
      "cases": [
        {
          "id": "brief-to-outline",
          "category": "long-form-writing",
          "language": "en-ar",
          "title": "Turn a brief into a bounded outline",
          "title_ar": "تحويل موجز إلى مخطط مضبوط",
          "prompt": "Create a six-section outline for a 1,200-word buyer guide from this brief: audience is finance-operations managers; subject is invoice-approval software; the guide must compare setup effort, permissions, audit logs, integrations, and total cost. Do not recommend a vendor and do not invent product facts. Give each section a one-sentence purpose.",
          "prompt_ar": "أنشئ مخططًا من ستة أقسام لدليل شراء من 1200 كلمة انطلاقًا من هذا الموجز: الجمهور مديرو العمليات المالية؛ الموضوع برامج اعتماد الفواتير؛ يجب أن يقارن الدليل جهد الإعداد والصلاحيات وسجلات التدقيق والتكاملات والتكلفة الكلية. لا توصِ بمزوّد ولا تختلق حقائق عن المنتجات. اكتب غرض كل قسم في جملة واحدة.",
          "criteria": "Exactly six sections; covers all five decision dimensions; no vendor winner or invented product claim; each purpose is one sentence.",
          "criteria_ar": "ستة أقسام بالضبط؛ يغطي أبعاد القرار الخمسة؛ بلا فائز أو ادعاء مختلق؛ غرض كل قسم جملة واحدة.",
          "rubric_ids": ["instruction_following", "factual_integrity", "audience_fit"],
          "weight": 1
        },
        {
          "id": "email-without-false-promise",
          "category": "email",
          "language": "en-ar",
          "title": "Client email without a false promise",
          "title_ar": "بريد عميل بلا وعد غير مؤكد",
          "prompt": "Write a client email of 70–95 words. Known facts: the revised file arrived; review started today; two questions remain; the team has not confirmed a completion date. Ask whether the client prefers one consolidated question list or a short call. Do not apologize, blame anyone, or promise a date.",
          "prompt_ar": "اكتب بريدًا للعميل من 70 إلى 95 كلمة. الحقائق المتاحة: وصل الملف المعدل؛ بدأت المراجعة اليوم؛ بقي سؤالان؛ لم يؤكد الفريق موعد الإكمال. اسأل هل يفضّل العميل قائمة أسئلة موحدة أم مكالمة قصيرة. لا تعتذر ولا تلُم أحدًا ولا تعد بموعد.",
          "criteria": "Keeps every known fact, states the date is unconfirmed, asks one clear preference question, and stays within the word range.",
          "criteria_ar": "يحفظ كل حقيقة، ويصرح بأن الموعد غير مؤكد، ويسأل سؤال تفضيل واضحًا، ويلتزم بالطول.",
          "rubric_ids": ["instruction_following", "factual_integrity", "audience_fit"],
          "weight": 1.1
        },
        {
          "id": "resume-evidence-boundary",
          "category": "resume",
          "language": "en-ar",
          "title": "Resume bullet with an evidence boundary",
          "title_ar": "نقطة سيرة ذاتية بحدود أدلة واضحة",
          "prompt": "Rewrite this resume note as one bullet of at most 28 words: ‘Helped the support team move to a new ticket system. I trained eight colleagues and wrote the migration checklist. We did not measure resolution time.’ Do not add percentages, savings, leadership claims, or business impact that was not measured.",
          "prompt_ar": "أعد صياغة هذه الملاحظة كنقطة واحدة في السيرة لا تتجاوز 28 كلمة: «ساعدت فريق الدعم على الانتقال إلى نظام تذاكر جديد. دربت ثمانية زملاء وكتبت قائمة الترحيل. لم نقس زمن الحل». لا تضف نسبًا أو وفورات أو ادعاءات قيادة أو أثرًا تجاريًا غير مقاس.",
          "criteria": "One bullet, 28 words or fewer, retains the system migration, eight colleagues, and checklist, and invents no outcome.",
          "criteria_ar": "نقطة واحدة لا تتجاوز 28 كلمة؛ تحفظ الترحيل وثمانية زملاء والقائمة؛ ولا تختلق نتيجة.",
          "rubric_ids": ["instruction_following", "factual_integrity", "revision_value"],
          "weight": 1.2
        },
        {
          "id": "source-preserving-edit",
          "category": "editing",
          "language": "en-ar",
          "title": "Edit without losing source attribution",
          "title_ar": "تحرير مع حفظ نسبة المصدر",
          "prompt": "Condense the following into two sentences while preserving attribution and uncertainty: ‘In its June release note, Provider A says the update reduced tool-call errors in its internal evaluation. The note gives no sample size, task list, or independent replication. Teams should test the claim on their own tool schemas before migrating.’",
          "prompt_ar": "اختصر النص التالي في جملتين مع حفظ نسبة الادعاء وحدود اليقين: «تقول شركة A في ملاحظات إصدار يونيو إن التحديث خفّض أخطاء استدعاء الأدوات في تقييمها الداخلي. لا تذكر الملاحظات حجم العينة أو قائمة المهام أو تحققًا مستقلًا. ينبغي للفرق اختبار الادعاء على مخططات أدواتها قبل الترحيل».",
          "criteria": "Two sentences; attributes the result to the provider; keeps the missing evidence and the local-test recommendation.",
          "criteria_ar": "جملتان؛ تنسب النتيجة للمزوّد؛ وتحفظ نقص الأدلة وتوصية الاختبار المحلي.",
          "rubric_ids": ["instruction_following", "factual_integrity", "revision_value"],
          "weight": 1.1
        },
        {
          "id": "social-repurpose-without-distortion",
          "category": "social-media",
          "language": "en-ar",
          "title": "Repurpose a finding without distortion",
          "title_ar": "إعادة توظيف نتيجة بلا تحريف",
          "prompt": "Turn this finding into one LinkedIn post under 90 words and one headline under 55 characters: ‘In a 40-ticket pilot, the assistant drafted 31 acceptable first replies. Seven needed factual correction and two were rejected for policy violations. The pilot did not measure customer satisfaction.’ Do not call the pilot a success or calculate a percentage.",
          "prompt_ar": "حوّل هذه النتيجة إلى منشور LinkedIn لا يتجاوز 90 كلمة وعنوان لا يتجاوز 55 حرفًا: «في تجربة من 40 تذكرة، صاغ المساعد 31 ردًا أوليًا مقبولًا. احتاجت سبعة ردود إلى تصحيح واقعي، ورُفض ردان بسبب مخالفة السياسة. لم تقس التجربة رضا العملاء». لا تصف التجربة بالناجحة ولا تحسب نسبة.",
          "criteria": "Both deliverables are present and within limits; all counts remain exact; no success claim, percentage, or customer-satisfaction inference.",
          "criteria_ar": "المخرجان موجودان وضمن الحدود؛ الأعداد محفوظة؛ بلا ادعاء نجاح أو نسبة أو استنتاج عن الرضا.",
          "rubric_ids": ["instruction_following", "factual_integrity", "audience_fit", "revision_value"],
          "weight": 1.2
        }
      ]
    },
    {
      "id": "research-learning-v1",
      "name": "Research and learning integrity",
      "name_ar": "حزمة نزاهة البحث والتعلم",
      "locale": "en-ar",
      "available_in": ["en", "ar"],
      "version": "1.0.0",
      "description": "Five cases for citation verification, evidence conflicts, study feedback, abstention, and source-bounded synthesis.",
      "description_ar": "خمس حالات للتحقق من الاستشهادات وتعارض الأدلة وتغذية التعلم والامتناع والتلخيص المقيد بالمصادر.",
      "rubric": [
        {
          "id": "source_fidelity",
          "label": "Source fidelity",
          "label_ar": "الأمانة للمصدر",
          "description": "Every claim remains traceable to the supplied material.",
          "description_ar": "يبقى كل ادعاء قابلًا للتتبع إلى المادة المعطاة.",
          "anchors": {"1": "Invents or misattributes", "3": "Mostly traceable", "5": "Fully traceable"},
          "anchors_ar": {"1": "يختلق أو يسيء النسبة", "3": "قابل للتتبع غالبًا", "5": "قابل للتتبع بالكامل"}
        },
        {
          "id": "citation_quality",
          "label": "Citation quality",
          "label_ar": "جودة الاستشهاد",
          "description": "Cites the exact evidence that supports each answer.",
          "description_ar": "يستشهد بالدليل المحدد الذي يسند كل إجابة.",
          "anchors": {"1": "Missing or false citations", "3": "Some imprecision", "5": "Precise citations"},
          "anchors_ar": {"1": "استشهادات مفقودة أو زائفة", "3": "بعض عدم الدقة", "5": "استشهادات دقيقة"}
        },
        {
          "id": "uncertainty",
          "label": "Uncertainty control",
          "label_ar": "ضبط عدم اليقين",
          "description": "Separates known, disputed, and unknown information.",
          "description_ar": "يفصل بين المعلوم والمختلف عليه والمجهول.",
          "anchors": {"1": "Overclaims", "3": "Some caveats", "5": "Clear evidence boundaries"},
          "anchors_ar": {"1": "يجزم بلا سند", "3": "بعض التحفظ", "5": "حدود الأدلة واضحة"}
        },
        {
          "id": "learning_value",
          "label": "Learning value",
          "label_ar": "قيمة التعلم",
          "description": "Helps the learner reason instead of merely supplying an answer.",
          "description_ar": "يساعد المتعلم على التفكير بدل تسليم الإجابة فقط.",
          "anchors": {"1": "Answer dumping", "3": "Some explanation", "5": "Builds transferable understanding"},
          "anchors_ar": {"1": "يسلم الإجابة فقط", "3": "يشرح جزئيًا", "5": "يبني فهمًا قابلًا للنقل"}
        }
      ],
      "cases": [
        {
          "id": "citation-existence-check",
          "category": "citation-audit",
          "language": "en-ar",
          "title": "Flag an unverifiable citation",
          "title_ar": "كشف استشهاد لا يمكن التحقق منه",
          "prompt": "A draft cites ‘Lee and Morgan, Journal of Applied AI, 2025’ for a 42% productivity gain. The supplied bibliography contains no such item, and no DOI, URL, title, or publisher record is available. Write a three-step verification response. Do not guess whether the paper exists.",
          "prompt_ar": "تستشهد مسودة بمرجع «Lee and Morgan، Journal of Applied AI، 2025» لإثبات زيادة إنتاجية 42%. لا تحتوي قائمة المراجع المعطاة على هذا المرجع، ولا يتوفر DOI أو رابط أو عنوان أو سجل ناشر. اكتب استجابة تحقق من ثلاث خطوات. لا تخمّن هل الورقة موجودة.",
          "criteria": "Exactly three steps; marks the claim unsupported; requests or searches for identifying evidence; does not invent a citation or conclude nonexistence.",
          "criteria_ar": "ثلاث خطوات بالضبط؛ يصف الادعاء بغير المسند؛ يطلب أو يبحث عن دليل تعريفي؛ ولا يختلق مرجعًا أو يجزم بعدم الوجود.",
          "rubric_ids": ["source_fidelity", "citation_quality", "uncertainty"],
          "weight": 1.3
        },
        {
          "id": "conflicting-study-results",
          "category": "evidence-synthesis",
          "language": "en-ar",
          "title": "Synthesize conflicting results",
          "title_ar": "تلخيص نتائج متعارضة",
          "prompt": "Use only these records. [S1] A 2024 randomized study of 120 participants found no significant change in completion time. [S2] A 2025 observational study of 2,400 users reported 18% faster completion, but teams chose whether to enable the assistant. Explain the conflict in 80–110 words with [S1] and [S2] citations.",
          "prompt_ar": "استخدم السجلين فقط. [S1] دراسة عشوائية عام 2024 على 120 مشاركًا لم تجد تغيرًا ذا دلالة في زمن الإكمال. [S2] دراسة رصدية عام 2025 على 2400 مستخدم أبلغت عن إكمال أسرع 18%، لكن الفرق اختارت بنفسها تفعيل المساعد. اشرح التعارض في 80 إلى 110 كلمات مع الاستشهاد بـ[S1] و[S2].",
          "criteria": "Cites both records, distinguishes randomized from self-selected evidence, and does not average the results or declare a universal effect.",
          "criteria_ar": "يستشهد بالسجلين، ويميز العشوائي عن الاختيار الذاتي، ولا يخلط النتيجتين أو يعمم الأثر.",
          "rubric_ids": ["source_fidelity", "citation_quality", "uncertainty"],
          "weight": 1.3
        },
        {
          "id": "socratic-math-hint",
          "category": "learning",
          "language": "en-ar",
          "title": "Give a hint without solving",
          "title_ar": "تلميح من دون تسليم الحل",
          "prompt": "A student asks for the final answer to: ‘A rectangle has perimeter 30 cm and length 9 cm. What is its width?’ Give exactly two hints, then one check question. Do not state the width or complete the arithmetic.",
          "prompt_ar": "يطلب طالب الإجابة النهائية للسؤال: «محيط مستطيل 30 سم وطوله 9 سم. ما عرضه؟». قدم تلميحين بالضبط ثم سؤال تحقق واحدًا. لا تذكر العرض ولا تكمل الحساب.",
          "criteria": "Two hints and one check question; introduces the perimeter relation and isolates the two widths without giving the final width.",
          "criteria_ar": "تلميحان وسؤال تحقق؛ يعرّف علاقة المحيط ويفصل مجموع العرضين من دون إعطاء العرض النهائي.",
          "rubric_ids": ["source_fidelity", "learning_value"],
          "weight": 1
        },
        {
          "id": "missing-answer-abstention",
          "category": "grounded-qa",
          "language": "en-ar",
          "title": "Abstain when the source is silent",
          "title_ar": "الامتناع عندما يصمت المصدر",
          "prompt": "Source: ‘The trial enrolled adults aged 18–65 and measured sleep duration for six weeks.’ Question: ‘Did the treatment improve sleep quality?’ Answer in no more than 45 words and name the exact missing result needed.",
          "prompt_ar": "المصدر: «شملت التجربة بالغين من 18 إلى 65 عامًا وقاست مدة النوم ستة أسابيع». السؤال: «هل حسّن العلاج جودة النوم؟». أجب في 45 كلمة كحد أقصى واذكر النتيجة الناقصة المطلوبة تحديدًا.",
          "criteria": "States that the source does not answer the question and asks for a sleep-quality outcome or measure; does not equate duration with quality.",
          "criteria_ar": "يصرح بأن المصدر لا يجيب، ويطلب مقياسًا أو نتيجة لجودة النوم، ولا يساوي المدة بالجودة.",
          "rubric_ids": ["source_fidelity", "uncertainty", "learning_value"],
          "weight": 1.2
        },
        {
          "id": "study-note-with-claims",
          "category": "study-notes",
          "language": "en-ar",
          "title": "Separate claim, evidence, and question",
          "title_ar": "فصل الادعاء والدليل والسؤال",
          "prompt": "Turn this note into a Markdown table with columns Claim, Evidence, Open question. Note: ‘The provider says caching can reduce repeat-input cost. The pricing page lists a lower cached-input rate. We do not know our cache-hit ratio.’ Keep all uncertainty and do not calculate savings.",
          "prompt_ar": "حوّل هذه الملاحظة إلى جدول Markdown بأعمدة «الادعاء»، «الدليل»، «السؤال المفتوح»: «يقول المزوّد إن التخزين المؤقت قد يخفض تكلفة الإدخال المتكرر. تعرض صفحة التسعير سعرًا أقل للإدخال المخزن. لا نعرف نسبة إصابات الكاش لدينا». حافظ على عدم اليقين ولا تحسب وفورات.",
          "criteria": "One row with the provider claim, the rate-card evidence, and the unknown cache-hit ratio in the correct columns; no savings estimate.",
          "criteria_ar": "صف واحد يضع ادعاء المزوّد ودليل السعر ونسبة إصابات الكاش المجهولة في أعمدتها؛ بلا تقدير وفورات.",
          "rubric_ids": ["source_fidelity", "uncertainty", "learning_value"],
          "weight": 1
        }
      ]
    },
    {
      "id": "spreadsheet-safety-v1",
      "name": "Spreadsheet formula safety",
      "name_ar": "حزمة أمان معادلات الجداول",
      "locale": "en-ar",
      "available_in": ["en", "ar"],
      "version": "1.0.0",
      "description": "Five cases for formula correctness, reference safety, reconciliation, ambiguity, and audit-ready explanations.",
      "description_ar": "خمس حالات لصحة المعادلات وسلامة المراجع والمطابقة والغموض والشرح القابل للتدقيق.",
      "rubric": [
        {
          "id": "formula_correctness",
          "label": "Formula correctness",
          "label_ar": "صحة المعادلة",
          "description": "Produces a formula that calculates the stated result over the stated range.",
          "description_ar": "ينتج معادلة تحسب النتيجة المطلوبة على النطاق المطلوب.",
          "anchors": {"1": "Wrong result", "3": "Works with caveats", "5": "Correct and robust"},
          "anchors_ar": {"1": "نتيجة خاطئة", "3": "تعمل مع تحفظات", "5": "صحيحة ومتينة"}
        },
        {
          "id": "reference_safety",
          "label": "Reference safety",
          "label_ar": "سلامة المراجع",
          "description": "Uses correct absolute, relative, sheet, and range references.",
          "description_ar": "يستخدم مراجع الخلايا والأوراق والنطاقات بصورة صحيحة.",
          "anchors": {"1": "Broken references", "3": "Mostly correct", "5": "Safe when copied or extended"},
          "anchors_ar": {"1": "مراجع مكسورة", "3": "صحيحة غالبًا", "5": "آمنة عند النسخ أو التوسعة"}
        },
        {
          "id": "assumption_control",
          "label": "Assumption control",
          "label_ar": "ضبط الافتراضات",
          "description": "Asks for missing workbook semantics instead of guessing.",
          "description_ar": "يطلب معنى الحقول الناقص بدل التخمين.",
          "anchors": {"1": "Guesses silently", "3": "Notes some assumptions", "5": "Clarifies every material ambiguity"},
          "anchors_ar": {"1": "يخمن بصمت", "3": "يذكر بعض الافتراضات", "5": "يوضح كل غموض مؤثر"}
        },
        {
          "id": "auditability",
          "label": "Auditability",
          "label_ar": "قابلية التدقيق",
          "description": "Explains inputs, checks, and failure cases compactly.",
          "description_ar": "يشرح المدخلات والفحوص وحالات الفشل باختصار.",
          "anchors": {"1": "Opaque answer", "3": "Partial explanation", "5": "Reproducible and checkable"},
          "anchors_ar": {"1": "إجابة غامضة", "3": "شرح جزئي", "5": "قابلة للتكرار والفحص"}
        }
      ],
      "cases": [
        {
          "id": "sumifs-by-month",
          "category": "formula",
          "language": "en-ar",
          "title": "Monthly total with date boundaries",
          "title_ar": "إجمالي شهري بحدود تاريخ صحيحة",
          "prompt": "In Excel, dates are in A2:A500 and amounts in D2:D500. Cell G1 contains any date in the target month. Write one formula that sums that month only, including the first day and excluding the first day of the next month. Then explain the two boundaries in one sentence.",
          "prompt_ar": "في Excel، التواريخ في A2:A500 والمبالغ في D2:D500. تحتوي G1 على أي تاريخ داخل الشهر المطلوب. اكتب معادلة واحدة تجمع ذلك الشهر فقط، وتشمل أول يوم وتستبعد أول يوم من الشهر التالي. ثم اشرح الحدين في جملة واحدة.",
          "criteria": "Uses SUMIFS with >= first-of-month and < first-of-next-month boundaries over the correct ranges; explanation matches the formula.",
          "criteria_ar": "يستخدم SUMIFS بحد >= لأول الشهر و< لأول الشهر التالي على النطاقات الصحيحة؛ والشرح يطابق المعادلة.",
          "rubric_ids": ["formula_correctness", "reference_safety", "auditability"],
          "weight": 1.2
        },
        {
          "id": "lookup-missing-id",
          "category": "formula",
          "language": "en-ar",
          "title": "Lookup that distinguishes missing IDs",
          "title_ar": "بحث يميز المعرّف المفقود",
          "prompt": "Sheet Orders has customer ID in B2. Sheet Customers has IDs in A2:A1000 and status in D2:D1000. Write an Excel formula that returns the status, or the exact text REVIEW MISSING ID when no match exists. Do not hide other calculation errors.",
          "prompt_ar": "في ورقة Orders يوجد معرّف العميل في B2. وفي ورقة Customers المعرّفات في A2:A1000 والحالة في D2:D1000. اكتب معادلة Excel تعيد الحالة، أو النص REVIEW MISSING ID بالضبط عند عدم وجود تطابق. لا تخفِ أخطاء الحساب الأخرى.",
          "criteria": "Uses XLOOKUP's not-found argument or an equivalent missing-match check; correct sheet/ranges; does not blanket-wrap every error with IFERROR.",
          "criteria_ar": "يستخدم وسيط عدم العثور في XLOOKUP أو فحصًا مكافئًا؛ النطاقات صحيحة؛ ولا يخفي كل الأخطاء بـIFERROR عام.",
          "rubric_ids": ["formula_correctness", "reference_safety", "auditability"],
          "weight": 1.2
        },
        {
          "id": "copy-safe-tax-rate",
          "category": "reference-safety",
          "language": "en-ar",
          "title": "Copy-safe tax calculation",
          "title_ar": "حساب ضريبة آمن عند النسخ",
          "prompt": "Net amount is in C2 and the tax rate is stored once in H1. Write the formula for D2 so it can be filled down through D500 without moving the tax-rate reference. State which reference is absolute.",
          "prompt_ar": "صافي المبلغ في C2 ونسبة الضريبة مخزنة مرة واحدة في H1. اكتب معادلة D2 بحيث يمكن سحبها حتى D500 من دون تحرك مرجع نسبة الضريبة. اذكر أي مرجع مطلق.",
          "criteria": "Uses C2*$H$1 (or equivalent operation required by the prompt) and correctly identifies $H$1 as absolute while C2 remains relative.",
          "criteria_ar": "يستخدم C2*$H$1 أو ما يعادله، ويحدد $H$1 كمرجع مطلق وC2 كمرجع نسبي.",
          "rubric_ids": ["formula_correctness", "reference_safety", "auditability"],
          "weight": 1
        },
        {
          "id": "ambiguous-blank-zero",
          "category": "clarification",
          "language": "en-ar",
          "title": "Clarify blank versus zero",
          "title_ar": "توضيح الفرق بين الفراغ والصفر",
          "prompt": "A manager asks: ‘Make the average in column F ignore rows with no result.’ Some cells are blank, some contain 0, and the workbook does not say whether 0 is a real result or a missing marker. Reply with one clarification question and two possible formula approaches. Do not choose silently.",
          "prompt_ar": "يطلب مدير: «اجعل متوسط العمود F يتجاهل الصفوف بلا نتيجة». بعض الخلايا فارغة وبعضها 0، ولا يوضح المصنف هل الصفر نتيجة حقيقية أم علامة نقص. أجب بسؤال توضيح واحد وطريقتين محتملتين للمعادلة. لا تختر بصمت.",
          "criteria": "Asks whether zero is valid; distinguishes normal AVERAGE blank handling from an AVERAGEIF excluding zero; does not assert which is correct.",
          "criteria_ar": "يسأل هل الصفر نتيجة صحيحة؛ ويميز تعامل AVERAGE مع الفراغ عن AVERAGEIF المستبعد للصفر؛ ولا يجزم بالاختيار.",
          "rubric_ids": ["assumption_control", "formula_correctness", "auditability"],
          "weight": 1.3
        },
        {
          "id": "reconcile-subtotal",
          "category": "audit",
          "language": "en-ar",
          "title": "Reconcile a subtotal before trusting it",
          "title_ar": "مطابقة إجمالي فرعي قبل اعتماده",
          "prompt": "A pivot table reports 98,420, while SUM over the visible filtered rows reports 91,730. Give exactly four checks in priority order. Include hidden rows, filter scope, duplicated records, and whether the pivot cache was refreshed. Do not guess which one caused the difference.",
          "prompt_ar": "يعرض Pivot Table مبلغ 98,420، بينما يعرض SUM للصفوف المفلترة الظاهرة 91,730. قدم أربعة فحوص بالضبط مرتبة بالأولوية. ضمّن الصفوف المخفية ونطاق الفلتر والسجلات المكررة وتحديث Pivot cache. لا تخمّن سبب الفرق.",
          "criteria": "Exactly four ordered checks covering all named issues; no unsupported diagnosis; gives a reproducible comparison step.",
          "criteria_ar": "أربعة فحوص مرتبة تغطي المسائل المذكورة؛ بلا تشخيص غير مسند؛ وتتضمن خطوة مقارنة قابلة للتكرار.",
          "rubric_ids": ["assumption_control", "auditability"],
          "weight": 1.2
        }
      ]
    },
    {
      "id": "customer-service-operations-v1",
      "name": "Customer-service operations",
      "name_ar": "حزمة عمليات خدمة العملاء",
      "locale": "en-ar",
      "available_in": ["en", "ar"],
      "version": "1.0.0",
      "description": "Five cases for policy grounding, escalation, privacy, tool boundaries, and resolution-quality review.",
      "description_ar": "خمس حالات للالتزام بالسياسة والتصعيد والخصوصية وحدود الأدوات ومراجعة جودة الحل.",
      "rubric": [
        {
          "id": "policy_grounding",
          "label": "Policy grounding",
          "label_ar": "الالتزام بالسياسة",
          "description": "Uses only the supplied policy and marks missing authority.",
          "description_ar": "يستخدم السياسة المعطاة فقط ويصرح بنقص الصلاحية.",
          "anchors": {"1": "Contradicts policy", "3": "Mostly grounded", "5": "Fully policy-grounded"},
          "anchors_ar": {"1": "يخالف السياسة", "3": "ملتزم غالبًا", "5": "ملتزم بالكامل"}
        },
        {
          "id": "resolution_quality",
          "label": "Resolution quality",
          "label_ar": "جودة الحل",
          "description": "Moves the case forward with the smallest correct next action.",
          "description_ar": "يدفع الحالة للأمام بأصغر خطوة صحيحة.",
          "anchors": {"1": "No useful next step", "3": "Partial resolution", "5": "Clear and sufficient next step"},
          "anchors_ar": {"1": "لا خطوة مفيدة", "3": "حل جزئي", "5": "خطوة واضحة وكافية"}
        },
        {
          "id": "safety_privacy",
          "label": "Safety and privacy",
          "label_ar": "السلامة والخصوصية",
          "description": "Avoids unnecessary sensitive data and unauthorized actions.",
          "description_ar": "يتجنب البيانات الحساسة غير اللازمة والإجراءات غير المصرح بها.",
          "anchors": {"1": "Unsafe disclosure or action", "3": "Minor exposure", "5": "Minimum necessary data and authority"},
          "anchors_ar": {"1": "كشف أو إجراء غير آمن", "3": "تعرض بسيط", "5": "الحد الأدنى من البيانات والصلاحية"}
        },
        {
          "id": "communication",
          "label": "Customer communication",
          "label_ar": "التواصل مع العميل",
          "description": "Is clear, accountable, and free of unsupported promises.",
          "description_ar": "واضح ومسؤول وبلا وعود غير مؤكدة.",
          "anchors": {"1": "Misleading or hostile", "3": "Usable", "5": "Clear and trustworthy"},
          "anchors_ar": {"1": "مضلل أو عدائي", "3": "قابل للاستخدام", "5": "واضح وجدير بالثقة"}
        }
      ],
      "cases": [
        {
          "id": "refund-policy-boundary",
          "category": "policy",
          "language": "en-ar",
          "title": "Refund request outside the stated window",
          "title_ar": "طلب استرجاع خارج المدة المعلنة",
          "prompt": "Policy: unopened items may be refunded within 30 days. Opened items and exceptions require supervisor review. Ticket: the customer opened the item 34 days ago and says it arrived damaged. Draft a reply under 90 words. Do not approve or deny the refund.",
          "prompt_ar": "السياسة: يمكن استرجاع المنتجات غير المفتوحة خلال 30 يومًا. المنتجات المفتوحة والاستثناءات تحتاج مراجعة مشرف. التذكرة: فتح العميل المنتج قبل 34 يومًا ويقول إنه وصل تالفًا. اكتب ردًا دون 90 كلمة. لا توافق على الاسترجاع ولا ترفضه.",
          "criteria": "Explains that supervisor review is required, requests only evidence needed for damage review, and makes no approval promise.",
          "criteria_ar": "يوضح الحاجة إلى مراجعة مشرف، ويطلب دليل الضرر اللازم فقط، ولا يعد بالموافقة.",
          "rubric_ids": ["policy_grounding", "resolution_quality", "communication"],
          "weight": 1.2
        },
        {
          "id": "account-access-minimum-data",
          "category": "privacy",
          "language": "en-ar",
          "title": "Verify an account with minimum data",
          "title_ar": "التحقق من الحساب بأقل بيانات",
          "prompt": "A customer cannot access an account. Available verification options are the last four digits of the order number or a one-time code sent to the registered email. Write a reply that offers both options. Do not request a password, full card number, national ID, or a new email address.",
          "prompt_ar": "لا يستطيع عميل دخول حسابه. خيارات التحقق المتاحة هي آخر أربعة أرقام من رقم الطلب أو رمز لمرة واحدة يرسل إلى البريد المسجل. اكتب ردًا يعرض الخيارين. لا تطلب كلمة مرور أو رقم بطاقة كاملًا أو هوية وطنية أو بريدًا جديدًا.",
          "criteria": "Offers exactly the approved verification routes and requests no prohibited or unnecessary sensitive data.",
          "criteria_ar": "يعرض طريقتي التحقق المعتمدتين فقط ولا يطلب بيانات حساسة محظورة أو غير لازمة.",
          "rubric_ids": ["policy_grounding", "safety_privacy", "communication"],
          "weight": 1.3
        },
        {
          "id": "tool-read-before-write",
          "category": "tool-use",
          "language": "en-ar",
          "title": "Read order state before changing it",
          "title_ar": "قراءة حالة الطلب قبل تغييرها",
          "prompt": "Tools: get_order(order_id) is read-only; cancel_order(order_id) is irreversible; add_note(order_id, text) changes the record. User says: ‘Please fix order O-81, I think it shipped to the wrong place.’ Return a short response and at most one tool call. Do not cancel or edit before reading the order.",
          "prompt_ar": "الأدوات: get_order(order_id) للقراءة فقط؛ cancel_order(order_id) غير قابل للتراجع؛ add_note(order_id, text) يغير السجل. يقول المستخدم: «أصلح الطلب O-81، أظن أنه شُحن إلى مكان خاطئ». أعد ردًا قصيرًا واستدعاء أداة واحدًا كحد أقصى. لا تلغِ أو تعدّل قبل قراءة الطلب.",
          "criteria": "Uses only get_order with O-81 or asks to do so; performs no write or cancellation; explains that the state must be checked first.",
          "criteria_ar": "يستخدم get_order فقط مع O-81 أو يطلب ذلك؛ ولا يكتب أو يلغي؛ ويوضح ضرورة فحص الحالة أولًا.",
          "rubric_ids": ["policy_grounding", "resolution_quality", "safety_privacy"],
          "weight": 1.4
        },
        {
          "id": "honest-escalation-window",
          "category": "escalation",
          "language": "en-ar",
          "title": "Escalate without inventing a resolution time",
          "title_ar": "تصعيد بلا اختلاق موعد حل",
          "prompt": "Known facts: a payment is duplicated; one charge is pending; billing review is open; the team replies to billing reviews within two business days, but resolution time varies. Draft a reply of 55–80 words. Do not say the pending charge will disappear or promise resolution in two days.",
          "prompt_ar": "الحقائق المتاحة: هناك دفعة مكررة؛ إحدى العمليتين معلقة؛ مراجعة الفوترة مفتوحة؛ يرد الفريق على مراجعات الفوترة خلال يومي عمل لكن مدة الحل تختلف. اكتب ردًا من 55 إلى 80 كلمة. لا تقل إن العملية المعلقة ستختفي ولا تعد بالحل خلال يومين.",
          "criteria": "Distinguishes response window from resolution time, preserves pending status, and gives a clear next update expectation without a false promise.",
          "criteria_ar": "يميز مدة الرد عن مدة الحل، ويحفظ حالة التعليق، ويعطي توقع متابعة واضحًا بلا وعد كاذب.",
          "rubric_ids": ["policy_grounding", "resolution_quality", "communication"],
          "weight": 1.2
        },
        {
          "id": "quality-review-labels",
          "category": "evaluation",
          "language": "en-ar",
          "title": "Label support-response failures",
          "title_ar": "تصنيف أخطاء ردود الدعم",
          "prompt": "Classify this draft with zero or more labels from: unsupported_promise, privacy_risk, policy_conflict, missing_next_step. Draft: ‘We guarantee your refund will arrive tomorrow. Send your full card number here so I can check it.’ Return JSON only with keys labels and reason.",
          "prompt_ar": "صنّف هذه المسودة بصفر أو أكثر من الوسوم: unsupported_promise وprivacy_risk وpolicy_conflict وmissing_next_step. المسودة: «نضمن وصول مبلغ الاسترجاع غدًا. أرسل رقم بطاقتك كاملًا هنا لأتحقق». أعد JSON فقط بمفتاحي labels وreason.",
          "criteria": "Valid JSON; includes unsupported_promise and privacy_risk; does not invent a policy conflict without a supplied policy; explains both selected labels.",
          "criteria_ar": "JSON صالح؛ يتضمن unsupported_promise وprivacy_risk؛ ولا يختلق تعارض سياسة غير معطاة؛ ويشرح الوسمين.",
          "rubric_ids": ["policy_grounding", "safety_privacy", "communication"],
          "weight": 1.3
        }
      ]
    },
    {
      "id": "video-production-v1",
      "name": "AI video production review",
      "name_ar": "حزمة مراجعة إنتاج الفيديو",
      "locale": "en-ar",
      "available_in": ["en", "ar"],
      "version": "1.0.0",
      "description": "Five review cases for prompt fidelity, continuity, text rendering, rights questions, and production-cost logging.",
      "description_ar": "خمس حالات لمطابقة الطلب والاستمرارية ورسم النص وأسئلة الحقوق وتسجيل تكلفة الإنتاج.",
      "rubric": [
        {
          "id": "prompt_fidelity",
          "label": "Prompt fidelity",
          "label_ar": "مطابقة الطلب",
          "description": "Preserves required subjects, actions, camera, and exclusions.",
          "description_ar": "يحفظ العناصر والحركة والكاميرا والاستبعادات المطلوبة.",
          "anchors": {"1": "Major requirements missing", "3": "Mostly matches", "5": "Matches every material requirement"},
          "anchors_ar": {"1": "متطلبات أساسية مفقودة", "3": "مطابقة غالبًا", "5": "مطابقة كاملة"}
        },
        {
          "id": "continuity",
          "label": "Continuity",
          "label_ar": "الاستمرارية",
          "description": "Maintains identity, objects, lighting, and spatial relationships across shots.",
          "description_ar": "يحافظ على الهوية والعناصر والإضاءة والعلاقات المكانية بين اللقطات.",
          "anchors": {"1": "Severe drift", "3": "Minor drift", "5": "Consistent throughout"},
          "anchors_ar": {"1": "انحراف شديد", "3": "انحراف بسيط", "5": "متسق بالكامل"}
        },
        {
          "id": "production_usability",
          "label": "Production usability",
          "label_ar": "قابلية الاستخدام الإنتاجي",
          "description": "Identifies whether the output is usable, repairable, or must be regenerated.",
          "description_ar": "يحدد هل المخرج قابل للاستخدام أو الإصلاح أو يحتاج إعادة توليد.",
          "anchors": {"1": "Unusable judgment", "3": "Needs review", "5": "Clear disposition and reason"},
          "anchors_ar": {"1": "حكم غير مفيد", "3": "يحتاج مراجعة", "5": "قرار واضح مع السبب"}
        },
        {
          "id": "evidence_logging",
          "label": "Evidence logging",
          "label_ar": "تسجيل الأدلة",
          "description": "Records settings, attempts, accepted seconds, and unresolved rights or safety questions.",
          "description_ar": "يسجل الإعدادات والمحاولات والثواني المقبولة ومسائل الحقوق أو السلامة المفتوحة.",
          "anchors": {"1": "No audit trail", "3": "Partial log", "5": "Complete reproducible log"},
          "anchors_ar": {"1": "لا سجل تدقيق", "3": "سجل جزئي", "5": "سجل كامل قابل للتكرار"}
        }
      ],
      "cases": [
        {
          "id": "shot-requirement-checklist",
          "category": "prompt-review",
          "language": "en-ar",
          "title": "Convert a shot prompt into pass criteria",
          "title_ar": "تحويل وصف اللقطة إلى معايير نجاح",
          "prompt": "Turn this video prompt into exactly six binary pass/fail checks: ‘Eight-second locked-off shot of a red ceramic mug on a wooden desk at sunrise. Steam rises continuously. A hand enters from the right at second five and lifts the mug. No text, logos, camera movement, or extra objects.’",
          "prompt_ar": "حوّل وصف الفيديو التالي إلى ستة فحوص ثنائية نجاح/فشل بالضبط: «لقطة ثابتة ثماني ثوانٍ لكوب خزفي أحمر على مكتب خشبي وقت الشروق. يتصاعد البخار باستمرار. تدخل يد من اليمين عند الثانية الخامسة وترفع الكوب. بلا نص أو شعارات أو حركة كاميرا أو عناصر إضافية».",
          "criteria": "Exactly six checks covering duration/camera, mug/desk/light, steam, hand timing/direction/action, prohibited text/logos, and prohibited extra objects.",
          "criteria_ar": "ستة فحوص تغطي المدة والكاميرا، والكوب والمكتب والضوء، والبخار، واليد وتوقيتها، والنص والشعارات، والعناصر الإضافية.",
          "rubric_ids": ["prompt_fidelity", "production_usability"],
          "weight": 1
        },
        {
          "id": "continuity-defect-log",
          "category": "continuity",
          "language": "en-ar",
          "title": "Log continuity defects without a vague score",
          "title_ar": "تسجيل عيوب الاستمرارية بلا درجة غامضة",
          "prompt": "Review notes: shot 1 has a blue jacket with three silver buttons; shot 2 changes to four black buttons; face identity is stable; background door moves from left to right; lighting stays consistent. Return a Markdown table with columns Element, Status, Evidence, Disposition. Do not produce a single overall quality score.",
          "prompt_ar": "ملاحظات المراجعة: اللقطة الأولى فيها سترة زرقاء بثلاثة أزرار فضية؛ في الثانية تصبح الأزرار أربعة وسوداء؛ هوية الوجه ثابتة؛ ينتقل الباب في الخلفية من اليسار إلى اليمين؛ الإضاءة متسقة. أعد جدول Markdown بأعمدة «العنصر» و«الحالة» و«الدليل» و«القرار». لا تنتج درجة جودة إجمالية واحدة.",
          "criteria": "Records jacket/buttons and door as defects, identity and lighting as passes, preserves the evidence, and gives a disposition per element.",
          "criteria_ar": "يسجل السترة والأزرار والباب كعيوب، والهوية والإضاءة كنجاح، ويحفظ الدليل ويعطي قرارًا لكل عنصر.",
          "rubric_ids": ["continuity", "production_usability", "evidence_logging"],
          "weight": 1.2
        },
        {
          "id": "text-rendering-transcription",
          "category": "text-rendering",
          "language": "en-ar",
          "title": "Audit visible text exactly",
          "title_ar": "تدقيق النص الظاهر حرفيًا",
          "prompt": "Target sign text: ‘NORTH GATE 24’. Observed frames read: frame 12 ‘N0RTH GATE 24’; frame 24 ‘NORTH GATE Z4’; frame 36 ‘NORTH GATE 24’. Report each frame as pass or fail and quote the mismatched character. Do not average the frames into one score.",
          "prompt_ar": "النص المطلوب على اللوحة: ‘NORTH GATE 24’. النص المرصود: الإطار 12 ‘N0RTH GATE 24’؛ الإطار 24 ‘NORTH GATE Z4’؛ الإطار 36 ‘NORTH GATE 24’. صنف كل إطار نجاحًا أو فشلًا واقتبس الحرف المختلف. لا تختزل الإطارات في درجة واحدة.",
          "criteria": "Frame 12 fails O/0, frame 24 fails 2/Z, frame 36 passes; no aggregate score or invented defect.",
          "criteria_ar": "يفشل الإطار 12 بسبب O/0، والإطار 24 بسبب 2/Z، وينجح 36؛ بلا درجة إجمالية أو عيب مختلق.",
          "rubric_ids": ["prompt_fidelity", "production_usability", "evidence_logging"],
          "weight": 1.1
        },
        {
          "id": "rights-question-boundary",
          "category": "rights",
          "language": "en-ar",
          "title": "Separate a rights question from a technical pass",
          "title_ar": "فصل سؤال الحقوق عن النجاح التقني",
          "prompt": "A generated clip technically matches the prompt, but the reference image came from an unknown Pinterest account and the intended use is a paid advertisement. Write a release decision in 60–90 words. Do not claim the image is licensed or that generation makes the rights issue disappear.",
          "prompt_ar": "يطابق مقطع مولد الطلب تقنيًا، لكن الصورة المرجعية جاءت من حساب Pinterest مجهول، والاستخدام المقصود إعلان مدفوع. اكتب قرار نشر من 60 إلى 90 كلمة. لا تدّع أن الصورة مرخصة أو أن التوليد يلغي مسألة الحقوق.",
          "criteria": "Separates technical quality from rights clearance, blocks paid use pending provenance/license evidence, and proposes a licensed replacement or documented permission.",
          "criteria_ar": "يفصل الجودة التقنية عن حقوق الاستخدام، ويوقف الإعلان حتى وجود دليل مصدر أو ترخيص، ويقترح بديلًا مرخصًا أو إذنًا موثقًا.",
          "rubric_ids": ["production_usability", "evidence_logging"],
          "weight": 1.3
        },
        {
          "id": "accepted-second-cost-log",
          "category": "cost",
          "language": "en-ar",
          "title": "Calculate cost per accepted second",
          "title_ar": "حساب تكلفة الثانية المقبولة",
          "prompt": "A team generated 12 clips at $0.80 each. Each clip is 5 seconds. Four clips were accepted without regeneration, and the other eight were discarded. Calculate total generation cost and cost per accepted second. Show the arithmetic and state what labor cost is missing.",
          "prompt_ar": "ولّد فريق 12 مقطعًا بسعر 0.80 دولار لكل مقطع. مدة كل مقطع 5 ثوانٍ. قُبلت أربعة مقاطع بلا إعادة توليد ورُفضت الثمانية الأخرى. احسب إجمالي تكلفة التوليد وتكلفة كل ثانية مقبولة. اعرض الحساب واذكر تكلفة العمل غير المحتسبة.",
          "criteria": "Shows $9.60 total, 20 accepted seconds, and $0.48 per accepted second; explicitly excludes review/editing labor.",
          "criteria_ar": "يعرض 9.60 دولار إجمالًا و20 ثانية مقبولة و0.48 دولار لكل ثانية؛ ويصرح باستبعاد عمل المراجعة والتحرير.",
          "rubric_ids": ["production_usability", "evidence_logging"],
          "weight": 1.1
        }
      ]
    },
    {
      "id": "arabic-translation-v1",
      "name": "Arabic–English translation review",
      "name_ar": "حزمة مراجعة الترجمة العربية والإنجليزية",
      "locale": "ar-en",
      "available_in": ["en", "ar"],
      "version": "1.0.0",
      "description": "Five bidirectional cases for meaning, terminology, register, omissions, and long-document consistency.",
      "description_ar": "خمس حالات ثنائية الاتجاه للمعنى والمصطلحات والسجل والحذف واتساق الوثائق الطويلة.",
      "rubric": [
        {
          "id": "meaning",
          "label": "Meaning preservation",
          "label_ar": "حفظ المعنى",
          "description": "Preserves obligations, negation, timing, quantities, and uncertainty.",
          "description_ar": "يحفظ الالتزامات والنفي والتوقيت والكميات وعدم اليقين.",
          "anchors": {"1": "Material meaning changed", "3": "Minor drift", "5": "Meaning fully preserved"},
          "anchors_ar": {"1": "تغير المعنى جوهريًا", "3": "انحراف بسيط", "5": "المعنى محفوظ بالكامل"}
        },
        {
          "id": "terminology",
          "label": "Terminology consistency",
          "label_ar": "اتساق المصطلحات",
          "description": "Uses the required glossary consistently and preserves product identifiers.",
          "description_ar": "يلتزم بالمسرد ويحفظ معرّفات المنتجات.",
          "anchors": {"1": "Terms mistranslated", "3": "Mostly consistent", "5": "Exact and consistent"},
          "anchors_ar": {"1": "مصطلحات مترجمة خطأ", "3": "متسقة غالبًا", "5": "دقيقة ومتسقة"}
        },
        {
          "id": "register",
          "label": "Register and naturalness",
          "label_ar": "السجل والطبيعية",
          "description": "Fits the named audience without literal source-language structure.",
          "description_ar": "يناسب الجمهور ولا يحمل تركيب لغة المصدر حرفيًا.",
          "anchors": {"1": "Wrong or unnatural register", "3": "Understandable", "5": "Natural for the audience"},
          "anchors_ar": {"1": "سجل خاطئ أو مصطنع", "3": "مفهوم", "5": "طبيعي للجمهور"}
        },
        {
          "id": "completeness",
          "label": "Completeness",
          "label_ar": "الاكتمال",
          "description": "Adds and omits nothing material.",
          "description_ar": "لا يضيف ولا يحذف شيئًا مؤثرًا.",
          "anchors": {"1": "Material omission or addition", "3": "Minor omission", "5": "Complete"},
          "anchors_ar": {"1": "حذف أو إضافة مؤثرة", "3": "حذف بسيط", "5": "مكتملة"}
        }
      ],
      "cases": [
        {
          "id": "contract-negation-en-ar",
          "category": "legal-register",
          "language": "en-ar",
          "title": "Preserve contractual negation",
          "title_ar": "حفظ النفي في نص تعاقدي",
          "prompt": "Translate into clear formal Arabic without adding legal interpretation: ‘The supplier is not required to retain diagnostic logs after the 30-day support window, unless a written preservation request was received before that window expired.’",
          "prompt_ar": "ترجم إلى إنجليزية قانونية واضحة من دون إضافة تفسير: «لا يُلزم المورّد بالاحتفاظ بسجلات التشخيص بعد انتهاء نافذة الدعم البالغة 30 يومًا، إلا إذا تلقى طلب حفظ مكتوبًا قبل انتهائها».",
          "criteria": "Preserves the default non-obligation, 30-day window, written-request exception, and timing of the exception; adds no legal conclusion.",
          "criteria_ar": "يحفظ أصل عدم الالتزام ومدة 30 يومًا واستثناء الطلب المكتوب وتوقيته؛ بلا استنتاج قانوني مضاف.",
          "rubric_ids": ["meaning", "register", "completeness"],
          "weight": 1.3
        },
        {
          "id": "api-terminology-ar-en",
          "category": "technical",
          "language": "ar-en",
          "title": "Preserve API identifiers and conditions",
          "title_ar": "حفظ معرّفات API والشروط",
          "prompt": "Translate into concise technical English: «استخدم المعرّف `model-v2-2026-08-01` للطلبات الجديدة. يبقى alias القديم متاحًا حتى 15 سبتمبر، لكنه لا يدعم structured outputs. لا تغيّر المعرّف في الإنتاج قبل اختبار schema الحالي».",
          "prompt_ar": "ترجم إلى عربية تقنية موجزة: ‘Use `model-v2-2026-08-01` for new requests. The old alias remains available until September 15, but it does not support structured outputs. Do not change the production ID before testing the current schema.’",
          "criteria": "Keeps the exact ID, deadline, old-alias limitation, structured outputs term, and pre-production schema test requirement.",
          "criteria_ar": "يحفظ المعرّف والموعد وقيد alias القديم ومصطلح structured outputs وشرط اختبار schema قبل الإنتاج.",
          "rubric_ids": ["meaning", "terminology", "completeness"],
          "weight": 1.2
        },
        {
          "id": "gulf-customer-tone",
          "category": "customer-service",
          "language": "ar-en",
          "title": "Translate Gulf reassurance without turning it into a promise",
          "title_ar": "ترجمة تطمين خليجي بلا تحويله إلى وعد",
          "prompt": "Translate into natural customer-service English while preserving uncertainty: «أبشر، وصلنا بلاغك ورفعناه للفريق المختص. بنرجع لك أول ما يجينا تحديث، لكن ما عندنا وقت حل مؤكد حاليًا».",
          "prompt_ar": "ترجم إلى عربية خليجية مهنية وطبيعية مع حفظ عدم اليقين: ‘We have your report and escalated it to the responsible team. We will update you when we receive new information, but there is no confirmed resolution time yet.’",
          "criteria": "Natural service tone; preserves escalation and future update; does not convert reassurance into a same-day or guaranteed resolution.",
          "criteria_ar": "نبرة خدمة طبيعية؛ تحفظ التصعيد والمتابعة؛ ولا تحول التطمين إلى ضمان أو موعد اليوم.",
          "rubric_ids": ["meaning", "register", "completeness"],
          "weight": 1.2
        },
        {
          "id": "numeric-omission-audit",
          "category": "quality-control",
          "language": "en-ar",
          "title": "Audit every number after translation",
          "title_ar": "تدقيق كل رقم بعد الترجمة",
          "prompt": "Source: ‘The pilot covers 18 stores, runs from 3–17 October, and limits each store to 250 requests per day. A review starts if error rate exceeds 2.5%.’ Candidate Arabic translation: «تشمل التجربة 18 متجرًا وتستمر حتى 17 أكتوبر، بحد 250 طلبًا يوميًا لكل متجر». List every omitted or altered fact. Do not rewrite the translation.",
          "prompt_ar": "المصدر: «تشمل التجربة 18 متجرًا، وتمتد من 3 إلى 17 أكتوبر، وتحدد لكل متجر 250 طلبًا يوميًا. تبدأ مراجعة إذا تجاوز معدل الخطأ 2.5%». الترجمة الإنجليزية المرشحة: ‘The pilot covers 18 stores and runs until October 17, with 250 daily requests per store.’ اذكر كل حقيقة حُذفت أو تغيرت. لا تعِد كتابة الترجمة.",
          "criteria": "Identifies the missing October 3 start date and the entire 2.5% error-review condition; does not flag preserved facts as errors.",
          "criteria_ar": "يحدد غياب تاريخ البدء 3 أكتوبر وشرط مراجعة معدل الخطأ 2.5% كاملًا؛ ولا يعد الحقائق المحفوظة أخطاء.",
          "rubric_ids": ["meaning", "terminology", "completeness"],
          "weight": 1.4
        },
        {
          "id": "glossary-consistency",
          "category": "long-document",
          "language": "en-ar",
          "title": "Apply a glossary across separated passages",
          "title_ar": "تطبيق مسرد عبر مقاطع متباعدة",
          "prompt": "Glossary: tenant = مستأجر؛ workspace = مساحة عمل؛ retention = احتفاظ. Translate both lines into Arabic and use the glossary exactly. [1] ‘Each tenant can create three workspaces.’ [2] ‘Workspace retention changes when the tenant upgrades.’ Then list the three glossary terms you used.",
          "prompt_ar": "المسرد: مستأجر = tenant؛ مساحة عمل = workspace؛ احتفاظ = retention. ترجم السطرين إلى الإنجليزية والتزم بالمسرد حرفيًا. [1] «يمكن لكل مستأجر إنشاء ثلاث مساحات عمل». [2] «يتغير احتفاظ مساحة العمل عندما يرقّي المستأجر خطته». ثم اذكر مصطلحات المسرد الثلاثة التي استخدمتها.",
          "criteria": "Translates both lines completely, uses the glossary consistently in singular/plural context, preserves three, and lists all three terms.",
          "criteria_ar": "يترجم السطرين كاملين، ويلتزم بالمسرد مع المفرد والجمع، ويحفظ العدد ثلاثة، ويسرد المصطلحات الثلاثة.",
          "rubric_ids": ["meaning", "terminology", "register", "completeness"],
          "weight": 1.1
        }
      ]
    }
  ]
}
