{
 "entity": "glm-5-2",
 "entity_type": "model",
 "category": "Summarising",
 "suite": "suite-2026-08-tasks/summarising",
 "model": "z-ai/glm-5.2",
 "judges": [
  "anthropic/claude-sonnet-5",
  "google/gemini-3.1-pro-preview",
  "openai/gpt-5.6-terra"
 ],
 "judge_protocol": "judge-2026-08",
 "anchors": "solo-2026-08-a",
 "score": 91,
 "caps": 0,
 "cost_usd": 0.1293,
 "started_at": "2026-08-13T09:40:43.640Z",
 "finished_at": "2026-08-13T09:41:23.701Z",
 "tasks": [
  {
   "id": "su1",
   "name": "Meeting notes to actions",
   "prompt": "From these notes, list ONLY the action items, each with its owner, as a bulleted list. If an owner is not stated, write 'unassigned'. Notes: 'Priya said the March invoices still aren't reconciled. Tom will chase the supplier by Friday. We discussed the new CRM — nobody has looked at pricing yet. Sam is on leave next week. The board pack needs the updated cashflow before the 20th; Priya said she'd do it if she has the numbers.'",
   "output": "* Chase the supplier by Friday (Owner: Tom)\n* Update the cashflow for the board pack before the 20th (Owner: Priya)",
   "latency_ms": 8591,
   "cost": 0.0012699174,
   "task_score": 5,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 4,
     "violates": false,
     "reasoning": "Missed two action items: reconciling March invoices and reviewing CRM pricing (both unassigned). Incomplete list despite correct format for included items."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 5,
     "violates": false,
     "reasoning": "The response misses two clear action items from the notes (reconciling March invoices and looking at CRM pricing) which should have been listed as 'unassigned'."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 5,
     "violates": false,
     "reasoning": "Correctly identifies Tom’s and Priya’s stated commitments, but omits likely action items to reconcile March invoices and review CRM pricing, both of which should be listed as unassigned. Priya’s task is also conditional on receiving numbers"
    }
   ]
  },
  {
   "id": "su2",
   "name": "Five-bullet summary",
   "prompt": "Summarise in exactly 5 bullets, max 15 words each, readable in 20 seconds: 'Our Q3 revenue rose 12% to £1.4m, driven mainly by the new retail channel which contributed £310k. Margins slipped from 41% to 37% because of higher shipping costs and a one-off warehouse move. Headcount stayed flat at 34. The main risk into Q4 is the supplier contract, which expires in November and has not been renegotiated. Cash at bank is £620k.'",
   "output": "* Q3 revenue grew 12% to £1.4m, boosted by £310k from retail.\n* Margins fell to 37% due to higher shipping and warehouse relocation costs.\n* Employee headcount remained flat at 34 staff members.\n* Key Q4 risk is the unrenegotiated supplier contract expiring in November.\n* Cash reserves currently stand at £620k in the bank.",
   "latency_ms": 4666,
   "cost": 0.0043228,
   "task_score": 9,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise, exactly 5 bullets, all under 15 words, covers all key facts clearly."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, including exactly 5 bullets and under 15 words per bullet. It accurately and concisely captures all key information."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Meets the exact five-bullet and 15-word limits, with all major points accurately covered. Minor omissions: previous 41% margin and the warehouse move being one-off."
    }
   ]
  },
  {
   "id": "su3",
   "name": "Do not invent",
   "prompt": "Summarise this in under 50 words. If a fact is not present, do not supply it: 'The council approved the scheme on Tuesday. Three members voted against. Construction is expected to begin next year.' Then state on a separate line: 'Not stated: ' followed by one thing a reader would want to know that the text does not say.",
   "output": "The council approved the scheme on Tuesday, with three members voting against it. Construction is expected to begin next year.\n\nNot stated: what the scheme entails.",
   "latency_ms": 1997,
   "cost": 0.0019714,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise summary under 50 words, correct format, reasonable 'not stated' item. Minor: could be more concise but meets constraints well."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, providing an accurate summary under 50 words and correctly identifying a relevant missing detail on a separate line."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately preserves all stated facts, stays under 50 words, uses a separate required “Not stated:” line, and identifies a relevant missing detail without inventing information."
    }
   ]
  },
  {
   "id": "su4",
   "name": "One-sentence gist",
   "prompt": "Reduce to ONE sentence of no more than 25 words: 'After eighteen months of testing across four hospitals, the trial found that the new triage system reduced average waiting times by 22 minutes, although staff reported it added roughly ten minutes of administrative work per shift and two sites abandoned it before the trial ended.'",
   "output": "An 18-month trial found the new triage system reduced wait times by 22 minutes but increased administrative work and saw two sites abandon it.",
   "latency_ms": 34460,
   "cost": 0.00368458,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate one-sentence summary, 24 words, captures key facts; omits hospital count and admin-time specifics but concise and correct."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly condenses the original text into a single, highly readable sentence of exactly 24 words, capturing all key details while strictly adhering to the length constraint."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate one-sentence, 24-word summary preserving the main benefit, administrative burden, and early abandonment. It omits four hospitals and the roughly ten minutes per shift detail, slightly reducing completeness."
    }
   ]
  },
  {
   "id": "su5",
   "name": "Preserve the caveat",
   "prompt": "Summarise in under 60 words, and make sure the limitation survives: 'Early results suggest the drug reduces symptoms in about 60% of patients. However the sample was only 34 people, all under 40, and there was no control group, so the finding should be treated as preliminary.'",
   "output": "Early results indicate the drug reduces symptoms in about 60% of patients. However, this finding is strictly preliminary due to severe limitations: a small sample of just 34 people, all under 40, and no control group.",
   "latency_ms": 2461,
   "cost": 0.0028096,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise, preserves limitation, under 60 words. Minor stylistic redundancy but no factual or instruction issues."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly summarizes the text in under 60 words (36 words) while retaining all the crucial limitations mentioned in the original prompt. It is clear, concise, and accurate."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately preserves the 60% result and all key limitations: 34 participants, all under 40, and no control group. It clearly states the finding is preliminary and remains well under 60 words."
    }
   ]
  },
  {
   "id": "su6",
   "name": "Summarise a disagreement",
   "prompt": "In under 80 words, summarise the disagreement fairly, giving each side equal weight: 'Marketing wants to cut the price to win share, arguing volume will more than compensate. Finance says the last price cut in 2024 produced a 4% volume lift against an 11% margin loss, and would rather spend the same money on retention.'",
   "output": "Marketing proposes cutting prices to win market share, arguing that increased sales volume will more than compensate for lower margins. Finance opposes this, citing 2024 data where a similar cut caused an 11% margin loss for just a 4% volume gain. Instead, Finance recommends investing the same funds into customer retention.",
   "latency_ms": 6664,
   "cost": 0.00278558,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, balanced, concise summary within word limit (~50 words), fairly represents both sides with correct figures."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly summarizes the disagreement, accurately capturing both perspectives with equal weight. It is clear, concise, highly useful, and well under the 80-word limit."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurate, balanced summary gives both positions equal prominence, preserves the 2024 evidence and retention alternative, and stays well under 80 words."
    }
   ]
  },
  {
   "id": "su7",
   "name": "Extract the decision",
   "prompt": "From this thread, state in one line WHAT was decided and WHO decided it. If no decision was reached, say so. Thread: 'A: shall we move the launch to October? B: I'd rather not, the trade show is in September. A: agreed, but we can't ship without the packaging. C: packaging lands 2 Sept, confirmed by the printer this morning. A: fine, September stands then.'",
   "output": "A decided that the launch will remain in September.",
   "latency_ms": 1387,
   "cost": 0.0015084,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": null,
     "violates": null,
     "reasoning": "unparseable/failed"
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows the instructions, accurately identifying both WHAT was decided and WHO decided it in a single, concise line."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Accurately states the decision—launch remains in September—and identifies A as the person who confirmed it. It is concise and in one line."
    }
   ]
  },
  {
   "id": "su8",
   "name": "Long to short, no loss",
   "prompt": "Compress to under 40 words without losing any of the four numbers: 'The programme ran for 3 years, cost £2.1m, reached 14,000 participants, and reported a 19% improvement in the primary outcome measure compared with the comparison group.'",
   "output": "The 3-year programme cost £2.1m, reached 14,000 participants, and reported a 19% primary outcome improvement versus the comparison group.",
   "latency_ms": 2970,
   "cost": 0.0030652,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Retains all four numbers, accurate, concise, under 40 words (20), meets task instructions well."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response successfully compresses the text to 19 words (well under the 40-word limit) while retaining all four required numbers and the core meaning."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Preserves all four numbers and the comparison-group context in 18 words. Clear, accurate, and concise."
    }
   ]
  },
  {
   "id": "su9",
   "name": "Summary for a specific reader",
   "prompt": "Summarise this for a finance director who cares only about cost and risk, in under 70 words: 'The migration will move 40 services to the new platform over 9 months. The team estimates 3 months of dual-running, during which both platforms are paid for. There is a penalty clause if we exit the old contract before June. Engineers expect a 30% reduction in incidents afterwards.'",
   "output": "The 9-month migration of 40 services incurs dual-running costs for 3 months. Exiting the old contract before June triggers a financial penalty. Post-migration, a 30% reduction in incidents will lower operational risk and associated support costs.",
   "latency_ms": 2987,
   "cost": 0.00155732,
   "task_score": 8,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 8,
     "violates": false,
     "reasoning": "Accurate, concise, under 70 words, focuses on cost/risk as requested; minor omission of explicit cost figures but faithful to source."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly captures the cost and risk implications of the migration. It is highly concise, well under the 70-word limit, and tailored exactly to the finance director's priorities."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Accurately highlights migration duration, dual-running cost exposure, early-exit penalty, and reduced incident risk. It reasonably links fewer incidents to lower support costs, though that saving is inferred rather than explicitly stated."
    }
   ]
  },
  {
   "id": "su10",
   "name": "Bullet the risks only",
   "prompt": "List ONLY the risks as bullets, maximum 5, no solutions, no preamble: 'The supplier is reliable but single-source. Demand is growing 20% a year. Our warehouse lease ends in 14 months. The finance system is unsupported after next April. Staff turnover is low. Payment terms with our largest customer are 90 days.'",
   "output": "* Single-source supplier dependency\n* Rapid demand growth straining capacity\n* Warehouse lease expiration in 14 months\n* Finance system losing support after April\n* Cash flow vulnerability due to 90-day payment terms from the largest customer",
   "latency_ms": 7283,
   "cost": 0.00131512,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Accurate, concise bullet list of risks only, no solutions/preamble, within 5 items, correctly excludes low turnover as non-risk."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly extracts the five risks as bullet points, includes no preamble or solutions, and stays within the maximum limit."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "Exactly five risk-only bullets, with no preamble or solutions. It accurately captures all material risks stated and appropriately excludes low staff turnover as not a risk."
    }
   ]
  },
  {
   "id": "su11",
   "name": "Headline and standfirst",
   "prompt": "Write a headline (max 9 words) and a one-sentence standfirst (max 30 words) for: 'A study of 1,200 UK small businesses found that those which adopted at least one AI tool in 2025 grew revenue 8% faster on average, though the researchers could not establish whether the tools caused the growth or successful firms simply adopt tools sooner.'",
   "output": "**Headline:** UK Small Businesses Using AI See Faster Revenue Growth\n\n**Standfirst:** A study of 1,200 UK small businesses found AI adopters grew revenue 8% faster on average, though researchers could not establish if the tools caused the success.",
   "latency_ms": 12471,
   "cost": 0.00368146,
   "task_score": 9,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Meets word limits, accurate, concise, clear; minor omission of adoption-timing nuance but overall strong and usable."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions. The headline is exactly 9 words, and the standfirst is a single sentence of 27 words. The content is accurate, clear, and highly useful for a busy professional."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 8,
     "violates": false,
     "reasoning": "Meets both word limits and clearly conveys the finding and causation caveat. Minor omissions: it does not specify adoption of at least one AI tool or that adoption occurred in 2025."
    }
   ]
  },
  {
   "id": "su12",
   "name": "Say when there is nothing to summarise",
   "prompt": "Summarise the key findings of this text in under 40 words: 'Thanks — noted. Will pick this up Monday.' If there are no findings to summarise, say so plainly instead of padding.",
   "output": "There are no findings to summarise. The text is simply a brief acknowledgment that a task will be addressed on Monday.",
   "latency_ms": 3520,
   "cost": 0.00093644,
   "task_score": 10,
   "judges": [
    {
     "judge": "anthropic/claude-sonnet-5",
     "score": 9,
     "violates": false,
     "reasoning": "Correctly identifies no findings exist, concise, follows instructions well, useful and clear."
    },
    {
     "judge": "google/gemini-3.1-pro-preview",
     "score": 10,
     "violates": false,
     "reasoning": "The response perfectly follows all instructions, correctly identifying that there are no findings and stating this plainly within the word limit."
    },
    {
     "judge": "openai/gpt-5.6-terra",
     "score": 10,
     "violates": false,
     "reasoning": "It correctly identifies that the text contains no findings and states this plainly. The added context is accurate, concise, and remains well under 40 words."
    }
   ]
  }
 ]
}