{
 "entity": "deepseek-v4-pro",
 "entity_type": "model",
 "category": "Summarising",
 "suite": "suite-2026-08-tasks/summarising",
 "model": "deepseek/deepseek-v4-pro",
 "judges": [
  "anthropic/claude-sonnet-5",
  "google/gemini-3.1-pro-preview",
  "openai/gpt-5.6-terra"
 ],
 "judge_protocol": "judge-2026-08",
 "anchors": "solo-2026-08-a",
 "score": 95,
 "caps": 0,
 "cost_usd": 0.1359,
 "started_at": "2026-08-13T09:39:17.181Z",
 "finished_at": "2026-08-13T09:40:43.636Z",
 "tasks": [
  {
   "id": "su1",
   "name": "Meeting notes to actions",
   "prompt": "From these notes, list ONLY the action items, each with its owner, as a bulleted list. If an owner is not stated, write 'unassigned'. Notes: 'Priya said the March invoices still aren't reconciled. Tom will chase the supplier by Friday. We discussed the new CRM — nobody has looked at pricing yet. Sam is on leave next week. The board pack needs the updated cashflow before the 20th; Priya said she'd do it if she has the numbers.'",
   "output": "- Reconcile March invoices (unassigned)\n- Chase the supplier by Friday (Tom)\n- Look at pricing for the new CRM (unassigned)\n- Update cashflow for the board pack before the 20th (Priya)",
   "latency_ms": 17691,
   "cost": 0.00271256334,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, follows format, captures all action items with owners; minor nuance on Priya's conditional commitment omitted but acceptable."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly extracts all action items, assigns the correct owners (using 'unassigned' when appropriate), formats them as a bulleted list, and includes no extraneous text."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "All action items are correctly extracted, non-actions are excluded, and owners are accurately assigned or marked unassigned. The conditional cashflow commitment is appropriately attributed to Priya. Clear, concise bulleted format."
    }
   ]
  },
  {
   "id": "su2",
   "name": "Five-bullet summary",
   "prompt": "Summarise in exactly 5 bullets, max 15 words each, readable in 20 seconds: 'Our Q3 revenue rose 12% to £1.4m, driven mainly by the new retail channel which contributed £310k. Margins slipped from 41% to 37% because of higher shipping costs and a one-off warehouse move. Headcount stayed flat at 34. The main risk into Q4 is the supplier contract, which expires in November and has not been renegotiated. Cash at bank is £620k.'",
   "output": "- Revenue rose 12% to £1.4m, driven by new retail channel adding £310k.\n- Margins fell from 41% to 37% due to higher shipping costs and warehouse move.\n- Headcount remained unchanged at 34 employees.\n- Key Q4 risk: supplier contract expires in November, not yet renegotiated.\n- Cash at bank stands at £620k.",
   "latency_ms": 9977,
   "cost": 0.00060401664,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise, exactly 5 bullets, each under 15 words, covers all key facts clearly and readable quickly."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions. It provides exactly 5 bullet points, each under the 15-word limit, accurately capturing the key financial and operational metrics in a highly readable format."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Exactly five bullets, each under 15 words. Accurately captures all key financials, drivers, risks, and headcount in a concise, highly readable format."
    }
   ]
  },
  {
   "id": "su3",
   "name": "Do not invent",
   "prompt": "Summarise this in under 50 words. If a fact is not present, do not supply it: 'The council approved the scheme on Tuesday. Three members voted against. Construction is expected to begin next year.' Then state on a separate line: 'Not stated: ' followed by one thing a reader would want to know that the text does not say.",
   "output": "The council approved the scheme on Tuesday, with three members voting against. Construction is expected to begin next year.\n\nNot stated: The nature or purpose of the scheme.",
   "latency_ms": 8150,
   "cost": 0.00034025091,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise summary under 50 words, correct format, valid 'not stated' item not present in text."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions. It provides an accurate summary under 50 words without adding external facts, and correctly identifies a relevant missing detail on a separate line using the exact required phrasing."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurate under-50-word summary includes all stated facts without additions. Separate line correctly identifies a useful missing detail: the scheme’s nature or purpose."
    }
   ]
  },
  {
   "id": "su4",
   "name": "One-sentence gist",
   "prompt": "Reduce to ONE sentence of no more than 25 words: 'After eighteen months of testing across four hospitals, the trial found that the new triage system reduced average waiting times by 22 minutes, although staff reported it added roughly ten minutes of administrative work per shift and two sites abandoned it before the trial ended.'",
   "output": "An 18-month, four-hospital trial found new triage system cut wait times by 22 minutes, but added ten admin minutes per shift and two sites withdrew.",
   "latency_ms": 79636,
   "cost": 0.008401128,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, single sentence, exactly 25 words, concise and clear summary meeting the constraint."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 8,
     "violates": false,
     "reasoning": "The response successfully condenses the text into a single sentence of exactly 25 words. However, it omits the article 'the' before 'new triage system', making it slightly telegraphic."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Exactly 25 words and one sentence. It accurately preserves the trial duration, hospitals, wait-time reduction, added workload, and two-site abandonment, though it omits that withdrawal occurred before the trial ended."
    }
   ]
  },
  {
   "id": "su5",
   "name": "Preserve the caveat",
   "prompt": "Summarise in under 60 words, and make sure the limitation survives: 'Early results suggest the drug reduces symptoms in about 60% of patients. However the sample was only 34 people, all under 40, and there was no control group, so the finding should be treated as preliminary.'",
   "output": "In a small, uncontrolled study of 34 people under 40, the drug reduced symptoms in about 60% of patients. These early results are preliminary.",
   "latency_ms": 8993,
   "cost": 0.000468299685,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise summary under 60 words, preserves limitation (small sample, no control, preliminary)."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response is flawless. It perfectly summarizes the text in just 24 words, well under the 60-word limit, while accurately preserving all the crucial limitations (small sample, under 40, uncontrolled, preliminary)."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately preserves the 60% result and all key limitations: small sample, participants under 40, no control group, and preliminary status. It is concise, clear, and under 60 words."
    }
   ]
  },
  {
   "id": "su6",
   "name": "Summarise a disagreement",
   "prompt": "In under 80 words, summarise the disagreement fairly, giving each side equal weight: 'Marketing wants to cut the price to win share, arguing volume will more than compensate. Finance says the last price cut in 2024 produced a 4% volume lift against an 11% margin loss, and would rather spend the same money on retention.'",
   "output": "Marketing argues a price cut will win market share, with higher volume offsetting the lower margin. Finance disagrees, citing the 2024 cut that raised volume only 4% while margin fell 11%, and prefers spending the same money on customer retention instead.",
   "latency_ms": 7159,
   "cost": 0.0003502272,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, fair, balanced, concise, under 80 words. Minor stylistic room but no real flaws."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response accurately and fairly summarizes both sides of the disagreement in clear, concise language. It perfectly adheres to the word count constraint and provides equal weight to both perspectives."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurate, balanced summary that gives both positions comparable weight, preserves the key 2024 evidence and retention alternative, and stays well under 80 words."
    }
   ]
  },
  {
   "id": "su7",
   "name": "Extract the decision",
   "prompt": "From this thread, state in one line WHAT was decided and WHO decided it. If no decision was reached, say so. Thread: 'A: shall we move the launch to October? B: I'd rather not, the trade show is in September. A: agreed, but we can't ship without the packaging. C: packaging lands 2 Sept, confirmed by the printer this morning. A: fine, September stands then.'",
   "output": "The launch remains in September, decided by A.",
   "latency_ms": 9368,
   "cost": 0.0005519628,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 7,
     "violates": false,
     "reasoning": "Accurate summary of decision and decider; concise one-liner, though B's input contributed too, slightly understating collaborative aspect."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, accurately identifying the decision and the decision-maker in a single, concise line."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately states the decision (launch remains in September) and identifies A as the person who made the final decision. It is one concise line and fully follows the task."
    }
   ]
  },
  {
   "id": "su8",
   "name": "Long to short, no loss",
   "prompt": "Compress to under 40 words without losing any of the four numbers: 'The programme ran for 3 years, cost £2.1m, reached 14,000 participants, and reported a 19% improvement in the primary outcome measure compared with the comparison group.'",
   "output": "The 3-year programme cost £2.1m, reached 14,000 participants, and showed a 19% improvement in the primary outcome vs. comparison group.",
   "latency_ms": 19814,
   "cost": 0.00120547722,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Retains all four numbers, under 40 words, accurate and concise, minor slight rewording but faithful."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response successfully compresses the text to 20 words (well under the 40-word limit) while retaining all four required numbers and the core meaning."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Under 40 words and preserves all four numbers, duration, cost, reach, and comparative outcome improvement. Clear, concise, and faithful to the original."
    }
   ]
  },
  {
   "id": "su9",
   "name": "Summary for a specific reader",
   "prompt": "Summarise this for a finance director who cares only about cost and risk, in under 70 words: 'The migration will move 40 services to the new platform over 9 months. The team estimates 3 months of dual-running, during which both platforms are paid for. There is a penalty clause if we exit the old contract before June. Engineers expect a 30% reduction in incidents afterwards.'",
   "output": "Migrating over 9 months incurs 3 months of dual-running costs. An early exit penalty applies if we cancel before June. The primary risk is prolonged parallel spending. Post-migration, a 30% incident reduction lowers operational risk and unplanned cost, but timing must avoid penalties.",
   "latency_ms": 5310,
   "cost": 0.00017835,
   "task_score": 9,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise, under 70 words, focuses on cost/risk as requested; minor redundancy but solid summary."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "Perfectly tailored to the finance director's priorities, highlighting dual-running costs, penalty risks, and long-term savings. It is highly concise, accurate, and well under the 70-word limit."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 9,
     "violates": false,
     "reasoning": "Accurately highlights dual-running cost, early-exit penalty risk, and expected incident reduction. It is concise, finance-focused, and under 70 words. Minor limitation: it does not quantify the scale of migration, though that is less releva"
    }
   ]
  },
  {
   "id": "su10",
   "name": "Bullet the risks only",
   "prompt": "List ONLY the risks as bullets, maximum 5, no solutions, no preamble: 'The supplier is reliable but single-source. Demand is growing 20% a year. Our warehouse lease ends in 14 months. The finance system is unsupported after next April. Staff turnover is low. Payment terms with our largest customer are 90 days.'",
   "output": "- Dependency on a single supplier creates critical vulnerability if their capacity or business continuity fails.\n- Rapid 20% annual demand growth risks outpacing current supply and operational capacity.\n- Looming warehouse lease expiration in 14 months threatens fulfillment continuity if not renewed or relocated.\n- Unsupported finance system after April exposes financial operations to security flaws and critical failure.\n- 90-day payment terms with largest customer strain cash flow, especially under growth and upcoming operational transitions.",
   "latency_ms": 21053,
   "cost": 0.0002262,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise risk-only bullets, max 5, no solutions/preamble. Minor slight elaboration beyond strict bullet brevity but well within task intent."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions. It provides exactly five bullet points, correctly identifies the risks while ignoring the non-risk (low turnover), includes no preamble or solutions, and is highly clear and professional."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Exactly five risk-only bullets, with no preamble or solutions. Accurately identifies all material risks in the scenario and appropriately excludes low staff turnover as a risk."
    }
   ]
  },
  {
   "id": "su11",
   "name": "Headline and standfirst",
   "prompt": "Write a headline (max 9 words) and a one-sentence standfirst (max 30 words) for: 'A study of 1,200 UK small businesses found that those which adopted at least one AI tool in 2025 grew revenue 8% faster on average, though the researchers could not establish whether the tools caused the growth or successful firms simply adopt tools sooner.'",
   "output": "**Headline:**  \nAI adopters saw 8% faster revenue growth, study finds\n\n**Standfirst:**  \nA study of 1,200 UK small businesses found that those using AI tools in 2025 grew revenue 8% faster on average, but researchers could not prove causation.",
   "latency_ms": 22375,
   "cost": 0.00119318847,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Concise, accurate, within word limits, captures caveat about causation clearly and usefully."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, including the word count constraints (9 words for the headline, 27 words for the standfirst). It accurately and concisely summarizes the provided text."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Meets both length limits and accurately conveys the sample, UK small-business context, 2025 adoption, 8% association, and lack of proven causation. It omits the specific “at least one AI tool” detail, a minor loss of precision."
    }
   ]
  },
  {
   "id": "su12",
   "name": "Say when there is nothing to summarise",
   "prompt": "Summarise the key findings of this text in under 40 words: 'Thanks — noted. Will pick this up Monday.' If there are no findings to summarise, say so plainly instead of padding.",
   "output": "No findings to summarise. This is a brief acknowledgment and scheduling note.",
   "latency_ms": 11446,
   "cost": 0.000087087,
   "task_score": 9,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Correctly identifies no findings exist, concise, follows instructions well, accurate and clear."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows the instructions, accurately identifying that there are no findings to summarize and stating so plainly in well under 40 words."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Correctly identifies that the text contains no findings. It is concise and clear, though the second sentence adds minor explanatory padding after the requested plain statement."
    }
   ]
  }
 ]
}