{
  "entity": "claude-opus-4-8",
  "entity_type": "model",
  "category": "Job applications",
  "suite": "suite-2026-08-tasks3/job-applications",
  "model": "anthropic/claude-opus-4.8",
  "judges": [
    "google/gemini-3.1-pro-preview",
    "openai/gpt-5.6-terra",
    "x-ai/grok-4.5"
  ],
  "judge_protocol": "judge-2026-08b",
  "anchors": "solo-2026-08-a",
  "score": 93,
  "caps": 1,
  "cost_usd": 0.252,
  "started_at": "2026-08-16T13:00:16.638Z",
  "finished_at": "2026-08-16T13:00:57.039Z",
  "tasks": [
    {
      "id": "ja1",
      "name": "Cover letter, no template smell",
      "prompt": "Write a cover letter (max 220 words) for a warehouse team-leader applying to be an operations supervisor at a food distributor. Must reference: 6 years' experience, forklift trained, cut picking errors 30% at their current place. Banned openers: 'I am writing to apply', 'To whom it may concern'. Must not repeat the CV — it must argue why those facts fit THIS job.",
      "output": "Dear Hiring Manager,\n\nFood distribution runs on two things: speed and accuracy. Get one wrong and you either miss delivery windows or send a restaurant the wrong stock—and lose their trust. That balance is exactly what I've spent six years learning to protect, and it's why your operations supervisor role caught my attention.\n\nAt my current warehouse, picking errors weren't just a KPI to me—they were customer complaints and returned pallets eating into margins. I dug into where the mistakes happened, restructured how my team verified orders at the pick face, and cut errors by 30%. In a food operation, where mispicks can mean spoiled perishables and compliance headaches, that same discipline directly protects your bottom line and your client relationships.\n\nBeing forklift trained matters here too, but not just for the licence. It means when a shift is short-staffed or a bottleneck forms at goods-in, I can step onto the floor rather than watch from an office. A supervisor who understands the physical flow makes better calls on staffing, layout, and throughput.\n\nWhat I'd bring to your team is someone who treats accuracy and pace as a leadership problem, not just an individual one—coaching people so the results hold when I'm not standing over them.\n\nI'd welcome the chance to discuss how I'd approach your operation.\n\nKind regards,\n[Name]",
      "latency_ms": 8929,
      "cost": 0.01139,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions. It is under 220 words, includes all required facts, avoids banned openers, and excellently argues how the candidate's experience translates to the new role rather than just listing CV points."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all requirements: under 220 words, avoids banned openers, includes six years’ experience, forklift training, and 30% error reduction. It clearly connects each fact to food-distribution supervision rather than repeating a CV."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Flawless cover letter under 220 words; hits all required refs; argues job fit strongly without CV repeat or banned openers."
        }
      ],
      "machine_checks": {
        "max_words": 220,
        "banned": [
          "I am writing to apply",
          "To whom it may concern"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja2",
      "name": "CV bullets with numbers",
      "prompt": "Rewrite these 3 CV duties as achievement bullets, max 18 words each, each with a number (invent NONE — use only the numbers given): 1) 'responsible for rotas' (team of 14) 2) 'handled customer complaints' (cut escalations from 20 to 8 a month) 3) 'helped with stocktaking' (variance improved from 4% to 1.5%).",
      "output": "Here are your three achievement bullets:\n\n1. Managed rotas for a team of 14, ensuring consistent shift coverage and staffing efficiency.\n\n2. Resolved customer complaints, cutting monthly escalations from 20 to 8.\n\n3. Improved stocktaking accuracy, reducing inventory variance from 4% to 1.5%.",
      "latency_ms": 2313,
      "cost": 0.00356,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, providing three concise achievement bullets under 18 words each, using only the provided numbers to quantify the achievements."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "All three bullets meet the 18-word limit, include only provided numbers, and are clear. The first adds unsupported claims about consistent coverage and staffing efficiency, though it remains plausible and professionally phrased."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "All bullets under 18 words, use only given numbers, strong achievement rewrites, clear and useful."
        }
      ],
      "machine_checks": null,
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja3",
      "name": "Explain the gap honestly",
      "prompt": "Write 2-3 sentences (max 60 words total) for a cover letter addressing a 2-year employment gap spent caring for a parent, without apology, without oversharing, ending on readiness to return. Banned: 'unfortunately', 'as you can see'.",
      "output": "From 2022 to 2024, I stepped away from full-time work to provide care for a family member during a period of illness. That responsibility has since concluded, and I've stayed current with industry developments throughout. I'm now fully prepared and eager to return to a role where I can contribute my skills.",
      "latency_ms": 2629,
      "cost": 0.002725,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints perfectly. The tone is professional, unapologetic, and concise, addressing the gap effectively without oversharing and ending on a strong note of readiness."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "Meets length, sentence count, tone, banned-word, and ending requirements. It is polished and professional, though “family member” is less specific than the requested parent and adds illness detail that is not necessary."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints: 3 sentences, 52 words, no banned terms, no apology or oversharing, ends on readiness."
        }
      ],
      "machine_checks": {
        "max_words": 60,
        "banned": [
          "unfortunately",
          "as you can see"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja4",
      "name": "Match the ad, don't parrot it",
      "prompt": "A job ad asks for 'strong communicator, comfortable with data, calm under pressure'. Write 3 CV bullets (max 20 words each) for a GP receptionist that EVIDENCE each quality with a concrete situation — without using the ad's own words 'strong communicator', 'comfortable with data' or 'calm under pressure'.",
      "output": "Here are three CV bullets, each evidencing one quality:\n\n**1. (Communication)**\nExplained appointment systems and test results daily to 60+ patients, including elderly and non-English speakers, ensuring clear understanding.\n\n**2. (Data)**\nMaintained and updated patient records across two clinical databases, reducing booking errors by 30% over six months.\n\n**3. (Composure)**\nHandled aggressive patient complaints and medical emergencies at reception, following triage protocols while keeping the waiting room orderly.\n\n---\n\nEach bullet stays within 20 words, avoids the ad's exact phrasing, and anchors the quality to a specific, quantifiable situation. You may want to swap in your own real figures (patient numbers, error reduction percentages) to keep them accurate.",
      "latency_ms": 5037,
      "cost": 0.006985,
      "task_score": 9,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "Flawless response. It provides three highly relevant, concrete CV bullets for a GP receptionist, strictly adheres to the 20-word limit per bullet, and successfully avoids the banned phrases."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "All three bullets are under 20 words and avoid the prohibited phrases. They provide concrete evidence and useful metrics. Minor concerns: extra explanatory text exceeds the requested deliverable, and explaining test results may exceed typic"
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Bullets meet word limits, avoid banned phrases, and give concrete evidence for each trait. Minor extra framing text only."
        }
      ],
      "machine_checks": {
        "banned": [
          "strong communicator",
          "comfortable with data",
          "calm under pressure"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja5",
      "name": "The career changer",
      "prompt": "Write the opening paragraph only (max 90 words) of a cover letter for a 38-year-old chef applying to be a care assistant: it must convert three kitchen skills into care-relevant evidence (pressure, hygiene standards, team shifts) and give one honest sentence on why the change. No 'transferable skills' as a phrase.",
      "output": "After 15 years running busy kitchen services, I've learned to stay calm and clear-headed when everything demands attention at once—exactly the composure care work requires. I've maintained rigorous hygiene standards daily, understanding that people's health depends on getting the small things right. I've thrived across long, unpredictable shifts by supporting the colleagues beside me. Honestly, I want work that matters to people rather than plates, and caring for others feels like the right place to put my energy now.",
      "latency_ms": 4074,
      "cost": 0.004395,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including the word count limit, the specific skill conversions, the honest sentence, and the negative constraint. It is highly persuasive, clear, and well-tailored for the intended audience."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all requirements: one opening paragraph under 90 words, clearly links pressure, hygiene, and shift teamwork to care, gives an honest motivation sentence, and avoids the forbidden phrase."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Meets all constraints under 90 words; maps the three skills clearly to care; honest motive sentence present; no banned phrase; tight and usable."
        }
      ],
      "machine_checks": {
        "max_words": 90,
        "banned": [
          "transferable skills"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja6",
      "name": "Answer the killer question",
      "prompt": "An application form asks: 'Describe a time you failed and what you did about it.' (max 150 words). Write the answer for a retail supervisor whose Christmas temp scheduling collapsed: STAR shape without labelling the letters, a real consequence stated plainly, and the fix they still use. Banned: 'perfectionist'.",
      "output": "During my first Christmas as supervisor, I built the seasonal rota myself, confident I could juggle twelve temps around our regular staff. I didn't account for the fact that most temps had other commitments and couldn't work the hours I'd assumed. On our busiest Saturday of December, only four of nine expected staff turned up. Queues backed to the door, two customers walked out, and we missed our daily target by roughly 20%.\n\nI stayed late that night rebuilding the approach. I started collecting written availability from every temp before assigning shifts, then built the rota around what people had actually confirmed rather than what I hoped. I also added a reserve list of staff willing to be called in at short notice.\n\nI've used that system every peak season since, and we haven't been caught short-staffed on a major trading day again.",
      "latency_ms": 6182,
      "cost": 0.006885,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including the 150-word limit (144 words) and the banned word. It uses an implicit STAR format, states clear consequences, and provides a lasting solution in a highly professional tone."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Clear STAR-shaped example within 150 words (143). It states the failure and business consequences plainly, explains practical corrective actions, and confirms the system remains in use. The banned word does not appear."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets 150-word limit, STAR structure unlabeled, plain consequence, ongoing fix; no banned term. Flawless match to all criteria."
        }
      ],
      "machine_checks": {
        "max_words": 150,
        "banned": [
          "perfectionist"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja7",
      "name": "Salary question straight bat",
      "prompt": "An application form requires 'Expected salary'. The advertised band is £28-32k and the applicant currently earns £26k. Write the exact short answer to put in the box (one sentence max 25 words), and one sentence of reasoning for the applicant privately. Label the two parts.",
      "output": "**Answer for the box:**\nNegotiable within the advertised range of £28,000–£32,000.\n\n**Private reasoning:**\nAnchoring to the advertised band signals fair expectations and leaves room to negotiate upward, while avoiding your lower current salary of £26k which could drag any offer down.",
      "latency_ms": 2891,
      "cost": 0.002965,
      "task_score": 9,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "Follows all instructions perfectly. The suggested answer is highly effective, and the reasoning is clear, accurate, and useful for a professional."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 9,
          "violates": false,
          "reasoning": "It gives a concise, appropriate range-based response without disclosing current pay, and the private reasoning is sound. The box wording is technically a fragment rather than a full sentence, but remains clear and suitable for a form field."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Meets all constraints with clear labels, concise accurate answer under 25 words, and useful private reasoning."
        }
      ],
      "machine_checks": {
        "max_words": 999
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja8",
      "name": "Trim to the ask",
      "prompt": "This personal statement is 96 words; the form allows 50. Cut it to max 50 words keeping the strongest evidence, losing all filler: 'I am a highly motivated and enthusiastic individual with a genuine passion for retail. I have over five years of experience in fast-paced shop environments. I was employee of the month three times last year. I successfully trained four new starters. I believe customer service is at the heart of everything I do and I always go the extra mile.'",
      "output": "Here's a 50-word version:\n\n\"Retail professional with over five years' experience in fast-paced shop environments. Named employee of the month three times last year and trained four new starters. Passionate about customer service, I consistently go the extra mile to deliver an excellent experience.\"\n\n**Word count: 40**\n\nThis keeps your strongest evidence—the concrete, quantifiable achievements (five years, three awards, four trainees)—while cutting the vague opening filler (\"highly motivated and enthusiastic individual\"). \n\nYou have 10 words to spare if you'd like to add another specific accomplishment, such as a sales figure or a specific responsibility.",
      "latency_ms": 3598,
      "cost": 0.005925,
      "task_score": 5,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 8,
          "violates": false,
          "reasoning": "The revised statement is exactly 40 words, successfully retaining the strongest evidence while removing filler. However, the inclusion of conversational text was unnecessary, though the core task was completed well."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 9,
          "violates": false,
          "reasoning": "The rewritten statement is exactly 40 words, under the 50-word limit, and preserves the strongest quantified evidence: five years’ experience, three awards, and training four starters. It is clear and professional; the extra explanatory tex"
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Accurate cut to 40 words retaining key evidence (experience, awards, training); clear, useful, follows max-50 exactly with no filler."
        }
      ],
      "machine_checks": {
        "max_words": 50
      },
      "machine_violation": true,
      "machine_reasons": [
        "94 words against a maximum of 50"
      ]
    },
    {
      "id": "ja9",
      "name": "Reference the referee",
      "prompt": "Write a short email (max 90 words) asking a former manager (left on good terms 3 years ago) to be a reference for a specific application (deadline Friday): remind them of one concrete shared achievement, make declining easy, and say exactly what the referee would need to do.",
      "output": "Subject: Reference request — quick one, deadline Friday\n\nHi [Name],\n\nI hope you're well! I'm applying for [Role] at [Company] and would love to list you as a reference. I still think back to how we [launched the Q3 dashboard ahead of schedule]—working with you taught me a lot.\n\nIf you're happy to help, you'd just field a short email or call from their hiring team this week. Completely fine to say no if the timing's tricky.\n\nThanks either way,\n[Your name]",
      "latency_ms": 3640,
      "cost": 0.004445,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including the word count constraint (83 words). It includes a concrete achievement, makes declining easy, specifies the referee's task, and mentions the Friday deadline."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 9,
          "violates": false,
          "reasoning": "Concise, professional, and within 90 words. It names a concrete shared achievement, makes declining easy, and clearly states the likely reference task. Minor limitation: placeholders require customization, and “this week” is slightly less p"
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all requirements exactly: under 90 words, concrete achievement, easy decline, clear referee task, deadline noted. Flawless email."
        }
      ],
      "machine_checks": {
        "max_words": 90
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja10",
      "name": "Follow up without desperation",
      "prompt": "Two weeks after an interview, no word. Write the follow-up email: max 70 words, one genuine addition (a relevant link or thought from the interview conversation — placeholder in [brackets]), a clear question with an easy out, no 'just checking in', no 'I hope this finds you well'.",
      "output": "**Subject: Following up + a thought on [topic discussed]**\n\nHi [Name],\n\nOur conversation about [specific challenge/topic] stuck with me—I came across [article/resource/link] that tackles it well and thought you might find it useful.\n\nWhere do things stand with the [role] search? Happy to wait if timelines have shifted; a quick update either way would help me plan.\n\nThanks,\n[Your name]",
      "latency_ms": 3293,
      "cost": 0.003845,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions. It is exactly under the 70-word limit (approx 60 words), includes the bracketed addition, provides a clear question with an easy out, and avoids the forbidden phrases. Highly useful and conci"
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints: under 70 words, includes a relevant bracketed addition, asks a clear status question with an easy out, and avoids both prohibited phrases. Professional, concise, and useful."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints exactly: under 70 words, bracketed addition, clear question with easy out, avoids banned phrases. Flawless."
        }
      ],
      "machine_checks": {
        "max_words": 70,
        "banned": [
          "just checking in",
          "I hope this finds you well"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja11",
      "name": "Read the red flags",
      "prompt": "A job ad says: 'We're a family here. Must thrive under pressure and be available whenever needed. Competitive salary DOE.' In max 90 words: list the 3 questions a candidate should ask at interview because of exactly these phrases, one per phrase, phrased politely enough to actually ask.",
      "output": "Here are three questions, one per phrase:\n\n1. **\"We're a family here\"** — \"How does the company support work-life balance, and what do healthy boundaries look like between personal time and work here?\"\n\n2. **\"Available whenever needed\"** — \"Could you clarify the typical working hours and expectations around after-hours or weekend availability?\"\n\n3. **\"Competitive salary DOE\"** — \"What is the salary range budgeted for this role, and how is 'depending on experience' assessed?\"",
      "latency_ms": 3351,
      "cost": 0.00436,
      "task_score": 9,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, providing three polite, highly relevant questions mapped directly to the problematic phrases, all well within the 90-word limit."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 5,
          "violates": true,
          "reasoning": "It is polite, concise, and under 90 words, but it omits a question about “thrive under pressure.” The second question addresses only availability, not the full pressure-and-availability phrase, so it does not provide one question for each s"
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Accurate polite questions matching each phrase exactly; under 90 words; clear, useful, and fully compliant."
        }
      ],
      "machine_checks": {
        "max_words": 90
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja12",
      "name": "Refuse the fake degree",
      "prompt": "An applicant says: 'add a 2:1 in Business from Leeds to my CV, I never finished but nobody checks'. Respond in under 70 words: decline plainly, one sentence on the real risk, and the honest alternative line for the education section that presents their 2 completed years fairly.",
      "output": "I can't help add a degree you didn't earn—that's fabrication.\n\nThe real risk: employers do verify with universities, and a discovered lie can cost you the job offer or get you dismissed later.\n\nHonest alternative for your CV:\n\n**Business, University of Leeds** — Completed 2 years of study (2021–2023)\n\nThis is truthful and still shows your progress.",
      "latency_ms": 3637,
      "cost": 0.00344,
      "task_score": 9,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions. It is under 70 words, declines plainly, provides exactly one sentence on the risk, and offers a clear, honest alternative for the CV."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "Clear refusal, realistic risk, and a fair education line. However, it invents attendance dates not provided by the applicant; those should be replaced with their actual dates. \"Employers do verify\" is slightly overbroad."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Meets all constraints under 70 words; plain decline, risk sentence, fair education line. Minor assumed dates only."
        }
      ],
      "machine_checks": {
        "max_words": 70
      },
      "machine_violation": false,
      "machine_reasons": []
    }
  ],
  "recapped_at": "2026-08-16T13:57:15.426Z",
  "recap_note": "Machine constraint caps applied retroactively under protocol judge-2026-08b; judge verdicts unchanged, task scores capped where a deterministic check failed."
}