{
 "entity": "qwen3-7-max",
 "entity_type": "model",
 "category": "Summarising",
 "suite": "suite-2026-08-tasks/summarising",
 "model": "qwen/qwen3.7-max",
 "judges": [
  "anthropic/claude-sonnet-5",
  "google/gemini-3.1-pro-preview",
  "openai/gpt-5.6-terra"
 ],
 "judge_protocol": "judge-2026-08",
 "anchors": "solo-2026-08-a",
 "score": 87,
 "caps": 1,
 "cost_usd": 0.2221,
 "started_at": "2026-08-13T09:47:06.995Z",
 "finished_at": "2026-08-13T09:49:37.821Z",
 "tasks": [
  {
   "id": "su1",
   "name": "Meeting notes to actions",
   "prompt": "From these notes, list ONLY the action items, each with its owner, as a bulleted list. If an owner is not stated, write 'unassigned'. Notes: 'Priya said the March invoices still aren't reconciled. Tom will chase the supplier by Friday. We discussed the new CRM — nobody has looked at pricing yet. Sam is on leave next week. The board pack needs the updated cashflow before the 20th; Priya said she'd do it if she has the numbers.'",
   "output": "* Reconcile March invoices - unassigned\n* Chase the supplier by Friday - Tom\n* Look at pricing for the new CRM - unassigned\n* Update cashflow for the board pack before the 20th - Priya",
   "latency_ms": 37779,
   "cost": 0.0090447,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, follows format, concise; minor nuance missed (Priya's conditional 'if she has the numbers') but overall correct and useful."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly extracts all action items and assigns the correct owners, including 'unassigned' where appropriate. It strictly follows the formatting constraint by providing ONLY the bulleted list."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately extracts all implied action items, assigns stated owners correctly, uses unassigned where no owner is given, and excludes non-action context. Concise bulleted format fully follows the task."
    }
   ]
  },
  {
   "id": "su2",
   "name": "Five-bullet summary",
   "prompt": "Summarise in exactly 5 bullets, max 15 words each, readable in 20 seconds: 'Our Q3 revenue rose 12% to £1.4m, driven mainly by the new retail channel which contributed £310k. Margins slipped from 41% to 37% because of higher shipping costs and a one-off warehouse move. Headcount stayed flat at 34. The main risk into Q4 is the supplier contract, which expires in November and has not been renegotiated. Cash at bank is £620k.'",
   "output": "* Q3 revenue rose 12% to £1.4m, driven by £310k from new retail.\n* Margins fell to 37% due to higher shipping and warehouse relocation costs.\n* Staff headcount remained steady at 34 employees.\n* Key Q4 risk is an unrenegotiated supplier contract expiring in November.\n* Current",
   "latency_ms": 70171,
   "cost": 0.017902075,
   "task_score": 3,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 3,
     "violates": true,
     "reasoning": "Response is truncated, missing final bullet content (cash £620k), failing to meet the 5-bullet requirement fully."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 3,
     "violates": false,
     "reasoning": "The response cuts off abruptly on the fifth bullet, omitting key financial information (cash at bank) and rendering the summary incomplete and unprofessional."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 5,
     "violates": false,
     "reasoning": "First four bullets accurately cover revenue, margins, headcount, and supplier risk. The fifth bullet is incomplete, omitting the £620k cash balance and reducing usability. It has exactly five bullets, each within 15 words."
    }
   ]
  },
  {
   "id": "su3",
   "name": "Do not invent",
   "prompt": "Summarise this in under 50 words. If a fact is not present, do not supply it: 'The council approved the scheme on Tuesday. Three members voted against. Construction is expected to begin next year.' Then state on a separate line: 'Not stated: ' followed by one thing a reader would want to know that the text does not say.",
   "output": "The council approved a scheme on Tuesday with three opposing votes; construction is expected next year.\nNot stated: The specific purpose of the scheme.",
   "latency_ms": 34100,
   "cost": 0.00765525,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise, under 50 words, follows format with 'Not stated' line and adds a reasonable unstated fact."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, providing an accurate summary well under the 50-word limit and correctly identifying a relevant missing detail on a separate line exactly as formatted."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately preserves all stated facts in a concise summary under 50 words. The separate line uses the required prefix and identifies a genuinely unstated, relevant detail without inventing information."
    }
   ]
  },
  {
   "id": "su4",
   "name": "One-sentence gist",
   "prompt": "Reduce to ONE sentence of no more than 25 words: 'After eighteen months of testing across four hospitals, the trial found that the new triage system reduced average waiting times by 22 minutes, although staff reported it added roughly ten minutes of administrative work per shift and two sites abandoned it before the trial ended.'",
   "output": "The new triage system reduced wait times by 22 minutes but added administrative work, prompting two of four trial hospitals to abandon it.",
   "latency_ms": 119735,
   "cost": 0.012243975,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "One sentence, 23 words, accurate summary though omits duration (18 months) and exact admin time; concise and clear."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly condenses the original text into a single, clear sentence of 23 words, capturing all key information while strictly adhering to the constraints."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Accurately captures the main benefit, administrative burden, and two withdrawals in one 23-word sentence. It slightly implies the extra work caused abandonment and omits the eighteen-month duration."
    }
   ]
  },
  {
   "id": "su5",
   "name": "Preserve the caveat",
   "prompt": "Summarise in under 60 words, and make sure the limitation survives: 'Early results suggest the drug reduces symptoms in about 60% of patients. However the sample was only 34 people, all under 40, and there was no control group, so the finding should be treated as preliminary.'",
   "output": "Early results suggest the drug reduces symptoms in about 60% of patients. However, findings remain preliminary due to key limitations: a small sample of 34 people, all under 40, and no control group.",
   "latency_ms": 44471,
   "cost": 0.01059935,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise, preserves limitation, under 60 words. Minor stylistic redundancy but meets task requirements well."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly summarizes the text in under 60 words (34 words) while retaining all the crucial limitations and maintaining clarity."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately preserves the 60% result and all stated limitations—small sample, age restriction, and no control group—while clearly retaining the preliminary conclusion. It is concise and under 60 words."
    }
   ]
  },
  {
   "id": "su6",
   "name": "Summarise a disagreement",
   "prompt": "In under 80 words, summarise the disagreement fairly, giving each side equal weight: 'Marketing wants to cut the price to win share, arguing volume will more than compensate. Finance says the last price cut in 2024 produced a 4% volume lift against an 11% margin loss, and would rather spend the same money on retention.'",
   "output": "Marketing advocates cutting prices to win share, arguing that projected volume surges will more than compensate for the resulting drop in profit margins. \n\nFinance opposes cutting prices, arguing that the 2024 4% volume lift failed to offset an 11% margin loss, and prefers spending that budget on retention.",
   "latency_ms": 41349,
   "cost": 0.010294025,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, balanced, concise; under 80 words, fair to both sides. Minor stylistic redundancy but no factual or instruction issues."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly summarizes the disagreement in under 80 words, giving exactly equal weight and space to both Marketing and Finance. It is clear, concise, and highly useful."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, balanced, and under 80 words. It fairly presents both positions and includes Finance’s supporting figures. “Projected volume surges” slightly overstates Marketing’s stated claim, which only says volume would compensate."
    }
   ]
  },
  {
   "id": "su7",
   "name": "Extract the decision",
   "prompt": "From this thread, state in one line WHAT was decided and WHO decided it. If no decision was reached, say so. Thread: 'A: shall we move the launch to October? B: I'd rather not, the trade show is in September. A: agreed, but we can't ship without the packaging. C: packaging lands 2 Sept, confirmed by the printer this morning. A: fine, September stands then.'",
   "output": "Person A decided to keep the launch in September.",
   "latency_ms": 19589,
   "cost": 0.0050386,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 5,
     "violates": false,
     "reasoning": "Reasonably captures decision and decider, but omits collaborative context (B and C's contributions) and slightly oversimplifies who 'decided'."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, stating in a single line exactly what was decided and who made the final decision based on the provided thread."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately identifies the decision: the launch remains in September, and attributes the final decision to A. It is concise, one line, and directly answers both required parts."
    }
   ]
  },
  {
   "id": "su8",
   "name": "Long to short, no loss",
   "prompt": "Compress to under 40 words without losing any of the four numbers: 'The programme ran for 3 years, cost £2.1m, reached 14,000 participants, and reported a 19% improvement in the primary outcome measure compared with the comparison group.'",
   "output": "Over 3 years, the £2.1m programme reached 14,000 participants, reporting a 19% primary outcome improvement versus the comparison group.",
   "latency_ms": 33221,
   "cost": 0.0089503,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Retains all four numbers, accurate, concise (19 words), clear and meets task requirements well."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly compresses the text to 19 words (well under the 40-word limit) while retaining all four required numbers and the original meaning."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately preserves all four numbers and the comparison-group context in 18 words. Clear, concise, and suitable for a non-technical reader."
    }
   ]
  },
  {
   "id": "su9",
   "name": "Summary for a specific reader",
   "prompt": "Summarise this for a finance director who cares only about cost and risk, in under 70 words: 'The migration will move 40 services to the new platform over 9 months. The team estimates 3 months of dual-running, during which both platforms are paid for. There is a penalty clause if we exit the old contract before June. Engineers expect a 30% reduction in incidents afterwards.'",
   "output": "**Cost:** Expect a temporary spike from three months of dual-platform billing. Long-term operational costs will decrease due to a projected 30% reduction in incidents. \n\n**Risk:** Financial penalties apply if the legacy contract is exited before June. Operational risk drops significantly post-migration. Ensure termination aligns with the June deadline to avoid unnecessary fees.",
   "latency_ms": 46632,
   "cost": 0.010255675,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise, focused on cost/risk, under 70 words, clear formatting; slightly more structured than plain summary but appropriate."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly targets the finance director by categorizing points into cost and risk. It captures all financial implications accurately and is well under the 70-word limit (54 words). Flawless execution."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Concise, finance-focused, and accurately identifies dual-running costs and early-exit penalty risk. It slightly overstates that incident reduction guarantees lower operational costs and omits the nine-month migration timeline, which may mat"
    }
   ]
  },
  {
   "id": "su10",
   "name": "Bullet the risks only",
   "prompt": "List ONLY the risks as bullets, maximum 5, no solutions, no preamble: 'The supplier is reliable but single-source. Demand is growing 20% a year. Our warehouse lease ends in 14 months. The finance system is unsupported after next April. Staff turnover is low. Payment terms with our largest customer are 90 days.'",
   "output": "- Single-source supplier dependency\n- Warehouse lease expiration in 14 months\n- Finance system becoming unsupported after next April\n- Cash flow constraints due to 90-day payment terms with the largest customer",
   "latency_ms": 23233,
   "cost": 0.005178725,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise, follows format, but omits demand-growth as a capacity risk; otherwise clean and compliant."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, providing exactly four bulleted risks with no preamble or solutions, accurately reflecting the provided text."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Follows the required bullet-only format, stays within five items, and identifies four clear risks. It omits the potential capacity/fulfilment risk from 20% annual demand growth, but otherwise is accurate and concise."
    }
   ]
  },
  {
   "id": "su11",
   "name": "Headline and standfirst",
   "prompt": "Write a headline (max 9 words) and a one-sentence standfirst (max 30 words) for: 'A study of 1,200 UK small businesses found that those which adopted at least one AI tool in 2025 grew revenue 8% faster on average, though the researchers could not establish whether the tools caused the growth or successful firms simply adopt tools sooner.'",
   "output": "**Headline:** AI linked to 8% faster small business growth\n\n**Standfirst:** A study found that UK small businesses adopting AI grew revenue 8% faster, though researchers cannot confirm whether the tools caused this growth or merely correlated with success.",
   "latency_ms": 60857,
   "cost": 0.014421075,
   "task_score": 9,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise, meets word limits (headline 9 words, standfirst 28 words), captures nuance of correlation vs causation."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, including word counts and sentence limits. The headline and standfirst are accurate, clear, and highly useful for a non-technical professional."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Meets word limits and clearly conveys the association and causation caveat. It accurately states the 8% revenue-growth finding, but omits the sample size, 2025 timing, and “at least one tool” detail."
    }
   ]
  },
  {
   "id": "su12",
   "name": "Say when there is nothing to summarise",
   "prompt": "Summarise the key findings of this text in under 40 words: 'Thanks — noted. Will pick this up Monday.' If there are no findings to summarise, say so plainly instead of padding.",
   "output": "There are no findings to summarise.",
   "latency_ms": 15020,
   "cost": 0.00318895,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Correctly identifies no findings exist, concise and follows instruction exactly, appropriate for non-technical audience."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows the instructions by plainly stating there are no findings to summarise, and it easily meets the word count constraint."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Correctly identifies that the text contains no substantive findings and states this plainly. It follows the under-40-word constraint and is concise, clear, and useful."
    }
   ]
  }
 ]
}