{
  "entity": "claude-fable-5",
  "entity_type": "model",
  "category": "Job applications",
  "suite": "suite-2026-08-tasks3/job-applications",
  "model": "anthropic/claude-fable-5",
  "judges": [
    "google/gemini-3.1-pro-preview",
    "openai/gpt-5.6-terra",
    "x-ai/grok-4.5"
  ],
  "judge_protocol": "judge-2026-08b",
  "anchors": "solo-2026-08-a",
  "score": 88,
  "caps": 2,
  "cost_usd": 0.3973,
  "started_at": "2026-08-16T12:58:27.965Z",
  "finished_at": "2026-08-16T12:59:28.186Z",
  "tasks": [
    {
      "id": "ja1",
      "name": "Cover letter, no template smell",
      "prompt": "Write a cover letter (max 220 words) for a warehouse team-leader applying to be an operations supervisor at a food distributor. Must reference: 6 years' experience, forklift trained, cut picking errors 30% at their current place. Banned openers: 'I am writing to apply', 'To whom it may concern'. Must not repeat the CV — it must argue why those facts fit THIS job.",
      "output": "Dear Hiring Manager,\n\nFood distribution runs on two things: speed and accuracy. Get either wrong and you're dealing with spoiled stock, missed delivery windows, or a customer switching suppliers. That's exactly the environment I've spent six years learning to manage — and why this operations supervisor role fits what I do best.\n\nAs a warehouse team leader, I cut picking errors by 30% by rebuilding our pick-path layout and introducing a simple two-stage verification at dispatch. In food distribution, that same discipline matters more, not less: a mispick isn't just a return, it's a broken cold chain or an allergen risk. I'd bring that error-reduction mindset to your operation from day one.\n\nMy forklift certification matters here for a practical reason: I don't supervise from an office. I can step onto the floor during peak receiving, cover a gap in the rota, and — more importantly — I understand what I'm asking of my team because I've done the work myself. That credibility is what makes safety standards stick.\n\nSix years leading shifts has taught me that supervising isn't about watching people work; it's about removing the obstacles that slow them down. I'd welcome the chance to discuss how I'd apply that at [Company Name].\n\nYours sincerely,\n[Name]",
      "latency_ms": 10068,
      "cost": 0.02673,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "Follows all constraints perfectly, including word count and banned phrases. Expertly translates CV facts into compelling arguments for the specific role, maintaining an engaging and professional tone throughout."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "Strong, tailored, concise letter that uses all key facts persuasively rather than listing the CV. Minor issue: it implies, rather than explicitly states, that the 30% reduction occurred at the current employer."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints exactly; argues fit to food ops role with required facts; clear, concise, professional and useful."
        }
      ],
      "machine_checks": {
        "max_words": 220,
        "banned": [
          "I am writing to apply",
          "To whom it may concern"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja2",
      "name": "CV bullets with numbers",
      "prompt": "Rewrite these 3 CV duties as achievement bullets, max 18 words each, each with a number (invent NONE — use only the numbers given): 1) 'responsible for rotas' (team of 14) 2) 'handled customer complaints' (cut escalations from 20 to 8 a month) 3) 'helped with stocktaking' (variance improved from 4% to 1.5%).",
      "output": "1. Scheduled and managed weekly rotas for a team of 14, ensuring full shift coverage.\n\n2. Resolved customer complaints at first contact, cutting monthly escalations from 20 to 8.\n\n3. Improved stocktaking accuracy, reducing inventory variance from 4% to 1.5%.",
      "latency_ms": 4657,
      "cost": 0.00802,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including the word count constraint and the requirement to use only the provided numbers. The bullets are concise, impactful, and highly useful."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "All three bullets are achievement-focused, use only provided numbers, remain within 18 words each, and are clear and useful. No explicit constraint is breached."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Perfect rewrite into achievements; all under 18 words; only given numbers used; clear and useful."
        }
      ],
      "machine_checks": null,
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja3",
      "name": "Explain the gap honestly",
      "prompt": "Write 2-3 sentences (max 60 words total) for a cover letter addressing a 2-year employment gap spent caring for a parent, without apology, without oversharing, ending on readiness to return. Banned: 'unfortunately', 'as you can see'.",
      "output": "From 2021 to 2023, I stepped away from my career to serve as my father's primary caregiver, a role that sharpened my organization, resilience, and ability to manage competing priorities under pressure. That chapter is now complete, and I'm energized and fully prepared to bring renewed focus and commitment to this role.",
      "latency_ms": 4986,
      "cost": 0.0078,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions. It is exactly two sentences, well under the 60-word limit (52 words), contains no banned phrases, avoids oversharing or apologizing, and ends with a strong statement of readiness."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all requirements: two sentences, under 60 words, directly explains the gap without apology or excessive detail, avoids both banned phrases, and ends with clear readiness to return."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets 2-sentence, 52-word limit; no banned terms; no apology/overshare; ends on readiness."
        }
      ],
      "machine_checks": {
        "max_words": 60,
        "banned": [
          "unfortunately",
          "as you can see"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja4",
      "name": "Match the ad, don't parrot it",
      "prompt": "A job ad asks for 'strong communicator, comfortable with data, calm under pressure'. Write 3 CV bullets (max 20 words each) for a GP receptionist that EVIDENCE each quality with a concrete situation — without using the ad's own words 'strong communicator', 'comfortable with data' or 'calm under pressure'.",
      "output": "**CV bullets — GP Receptionist**\n\n- Explained appointment changes and referral processes to 80+ patients daily, including elderly and non-English speakers, resolving most queries first-time.\n\n- Maintained and audited patient records on EMIS, spotting booking errors weekly and producing monthly DNA-rate reports for practice manager.\n\n- Triaged phone lines during flu-season surges, prioritising urgent cases and reassuring distressed callers while managing a full waiting room.\n\n**Why these work:**\n1. *Communication* — shown through volume, difficult audiences, and a measurable outcome (first-time resolution).\n2. *Data* — shown through a named system, error-spotting, and regular reporting.\n3. *Pressure* — shown through a high-stress scenario (surges, distressed callers, competing demands) handled with prioritisation.",
      "latency_ms": 8148,
      "cost": 0.01777,
      "task_score": 8,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 8,
          "violates": false,
          "reasoning": "The bullets perfectly evidence the required traits, meet the word count, and avoid banned phrases. However, the inclusion of unrequested introductory and explanatory text slightly detracts from concision."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "All three bullets meet the 20-word limit, avoid the forbidden phrases, and give concrete evidence for each quality. The extra explanatory section was not requested and slightly reduces concision."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Three strong bullets under 20 words each; concrete evidence without banned phrases. Minor extra section unneeded but harmless."
        }
      ],
      "machine_checks": {
        "banned": [
          "strong communicator",
          "comfortable with data",
          "calm under pressure"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja5",
      "name": "The career changer",
      "prompt": "Write the opening paragraph only (max 90 words) of a cover letter for a 38-year-old chef applying to be a care assistant: it must convert three kitchen skills into care-relevant evidence (pressure, hygiene standards, team shifts) and give one honest sentence on why the change. No 'transferable skills' as a phrase.",
      "output": "Dear Hiring Manager,\n\nAfter 20 years as a chef, I am applying to become a care assistant, bringing habits that matter in care: I stay calm and focused when demands peak, because dinner service teaches you to prioritise under real pressure; I hold rigorous hygiene and safety standards, having run kitchens to strict food safety compliance; and I work reliably across long, rotating shifts as part of a close team that depends on each other. Honestly, I want work where looking after people is the whole job, not a byproduct of it.",
      "latency_ms": 7021,
      "cost": 0.01319,
      "task_score": 5,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 5,
          "violates": true,
          "reasoning": "The response is 92 words long, violating the maximum 90-word constraint. Otherwise, it perfectly translates the three required skills, includes a compelling reason for the career change, and avoids the banned phrase."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "The 89-word opening paragraph clearly converts pressure, hygiene compliance, and team shift work into care-relevant evidence, includes an honest motivation sentence, avoids the banned phrase, and is concise and professional."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Meets all constraints exactly; converts skills well with honest reason; clear, concise, under 90 words; minor polish only possible."
        }
      ],
      "machine_checks": {
        "max_words": 90,
        "banned": [
          "transferable skills"
        ]
      },
      "machine_violation": true,
      "machine_reasons": [
        "92 words against a maximum of 90"
      ]
    },
    {
      "id": "ja6",
      "name": "Answer the killer question",
      "prompt": "An application form asks: 'Describe a time you failed and what you did about it.' (max 150 words). Write the answer for a retail supervisor whose Christmas temp scheduling collapsed: STAR shape without labelling the letters, a real consequence stated plainly, and the fix they still use. Banned: 'perfectionist'.",
      "output": "Last Christmas, I was responsible for scheduling fourteen temporary staff across our busiest trading fortnight. I built the rota entirely from the availability forms temps had submitted at induction, weeks earlier, and never checked back with anyone.\n\nOn the Saturday before Christmas, five temps didn't turn up. Their availability had changed and I hadn't asked. We ran two tills instead of four, queues stretched past the fitting rooms, and several customers abandoned full baskets and left. My manager had to come in on her day off.\n\nThat week I phoned every temp, rebuilt the rota around confirmed availability, and introduced a simple system: confirmation texts 48 hours before each shift, plus a paid on-call standby for peak days. Nobody unconfirmed goes on the rota.\n\nWe still run that system every peak season. I haven't had an uncovered shift since.",
      "latency_ms": 12003,
      "cost": 0.03042,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "Flawless response. It perfectly follows the STAR method without labels, clearly states a real consequence, includes the permanent fix, avoids the banned word, and stays under the 150-word limit (142 words)."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Clear STAR-shaped example within 150 words (137). It plainly states the operational and customer consequences, accepts responsibility, describes a practical fix, and confirms the system remains in use. The banned word is absent."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints exactly: under 150 words, STAR unlabelled, plain consequence, ongoing fix, no banned word. Clear, useful, accurate."
        }
      ],
      "machine_checks": {
        "max_words": 150,
        "banned": [
          "perfectionist"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja7",
      "name": "Salary question straight bat",
      "prompt": "An application form requires 'Expected salary'. The advertised band is £28-32k and the applicant currently earns £26k. Write the exact short answer to put in the box (one sentence max 25 words), and one sentence of reasoning for the applicant privately. Label the two parts.",
      "output": "**Answer for form:** \"£30,000–£32,000, negotiable depending on the overall package and responsibilities of the role.\"\n\n**Private reasoning:** Anchoring in the upper half of the advertised band positions you above your current £26k without exceeding their stated budget, while \"negotiable\" keeps flexibility if they offer £28-29k.",
      "latency_ms": 4987,
      "cost": 0.00798,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including sentence and word count constraints. The advice is highly practical, accurate, and clearly presented for a non-technical professional."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets the requested labeled two-part format. The form answer is one sentence under 25 words, realistic within the advertised band, and the private reasoning clearly explains the negotiation strategy."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Perfectly meets all specs: labeled parts, form answer under 25 words, one private sentence, accurate and useful advice."
        }
      ],
      "machine_checks": {
        "max_words": 999
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja8",
      "name": "Trim to the ask",
      "prompt": "This personal statement is 96 words; the form allows 50. Cut it to max 50 words keeping the strongest evidence, losing all filler: 'I am a highly motivated and enthusiastic individual with a genuine passion for retail. I have over five years of experience in fast-paced shop environments. I was employee of the month three times last year. I successfully trained four new starters. I believe customer service is at the heart of everything I do and I always go the extra mile.'",
      "output": "**Revised statement (31 words):**\n\n\"Retail professional with over five years' experience in fast-paced shop environments. Named employee of the month three times last year and successfully trained four new starters. Committed to delivering excellent customer service.\"\n\n**What was cut and why:**\n- \"Highly motivated and enthusiastic individual\" / \"genuine passion\" — unverifiable claims; the awards prove motivation better\n- \"I believe customer service is at the heart of everything I do\" — vague; condensed to one direct phrase\n- \"Go the extra mile\" — cliché with no evidence attached\n\n**What was kept:** the three concrete, verifiable facts (5 years' experience, 3× employee of the month, trained 4 starters), which do all the persuasive work. You have 19 words spare if you want to add a specific detail, e.g. the type of retail or a sales result.",
      "latency_ms": 9932,
      "cost": 0.02965,
      "task_score": 5,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 5,
          "violates": true,
          "reasoning": "The revised statement is excellent and under 50 words, but the total response exceeds the 50-word limit by including unnecessary explanations and commentary."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 9,
          "violates": false,
          "reasoning": "The revised statement keeps all three strongest, concrete achievements and is 32 words, within the 50-word limit. It mistakenly labels itself 31 words and adds unnecessary explanation, but the usable statement is accurate, concise, and stro"
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Accurate cut to 31 words retaining all key evidence; clear explanations; highly useful and polished with only trivial extras."
        }
      ],
      "machine_checks": {
        "max_words": 50
      },
      "machine_violation": true,
      "machine_reasons": [
        "129 words against a maximum of 50"
      ]
    },
    {
      "id": "ja9",
      "name": "Reference the referee",
      "prompt": "Write a short email (max 90 words) asking a former manager (left on good terms 3 years ago) to be a reference for a specific application (deadline Friday): remind them of one concrete shared achievement, make declining easy, and say exactly what the referee would need to do.",
      "output": "**Subject: Reference request — deadline Friday**\n\nHi Sarah,\n\nI hope you're well! I'm applying for a Senior Analyst role at Meridian Health and would love to list you as a reference — I still think of the CRM migration we delivered together three months ahead of schedule.\n\nIf you're willing, you'd just need to reply confirming, then complete a short online form (5–10 minutes) that Meridian will email you by Friday.\n\nNo pressure at all — if timing doesn't work, I completely understand.\n\nThanks so much,\nAlex",
      "latency_ms": 7367,
      "cost": 0.01194,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including the strict word count limit (86 words). It includes a concrete shared achievement, clearly outlines the referee's tasks, provides an easy out, and maintains a professional, warm ton"
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all requirements: concise, under 90 words, names the specific role and Friday deadline, cites a concrete shared achievement, makes declining easy, and clearly states the referee’s required actions."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 9,
          "violates": false,
          "reasoning": "Meets all requirements precisely under 90 words; clear, complete, professional email with achievement, easy decline, and exact referee steps."
        }
      ],
      "machine_checks": {
        "max_words": 90
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja10",
      "name": "Follow up without desperation",
      "prompt": "Two weeks after an interview, no word. Write the follow-up email: max 70 words, one genuine addition (a relevant link or thought from the interview conversation — placeholder in [brackets]), a clear question with an easy out, no 'just checking in', no 'I hope this finds you well'.",
      "output": "**Subject: Following up — [Role Title] interview**\n\nHi [Name],\n\nOur conversation about [specific topic discussed] stuck with me — I came across [link/article] afterward and thought it connected well to what your team is tackling.\n\nIs there an update on the timeline for a decision? If things are still in flux, no need for a detailed reply — a one-line status is plenty.\n\nThanks again for your time.\n\nBest,\n[Your Name]\n\n*(64 words)*",
      "latency_ms": 5705,
      "cost": 0.01089,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 5,
          "violates": true,
          "reasoning": "The response is well-written and meets all content requirements, but it violates the maximum 70-word constraint when counting all words provided (including the subject line and the word count note)."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints: under 70 words, includes a bracketed genuine addition, asks clearly for an update with an easy-out, and avoids both prohibited phrases. Professional, concise, and practical."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints exactly: under 70 words, bracketed addition, clear question with easy out, avoids banned phrases. Flawless and useful."
        }
      ],
      "machine_checks": {
        "max_words": 70,
        "banned": [
          "just checking in",
          "I hope this finds you well"
        ]
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja11",
      "name": "Read the red flags",
      "prompt": "A job ad says: 'We're a family here. Must thrive under pressure and be available whenever needed. Competitive salary DOE.' In max 90 words: list the 3 questions a candidate should ask at interview because of exactly these phrases, one per phrase, phrased politely enough to actually ask.",
      "output": "**1. \"We're a family here\":** \"Could you describe how the team handles disagreements or setting boundaries—for example, how work-life separation is respected?\"\n\n**2. \"Available whenever needed / thrive under pressure\":** \"Can you walk me through typical working hours and how often evenings, weekends, or on-call availability are actually expected?\"\n\n**3. \"Competitive salary DOE\":** \"Could you share the budgeted salary range for this role so we can confirm we're aligned before moving forward?\"\n\nEach question politely surfaces the risk hidden in the phrase: blurred boundaries, unpaid overtime, and undefined pay.",
      "latency_ms": 6166,
      "cost": 0.01407,
      "task_score": 8,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions, including the 90-word limit. It provides three polite, highly relevant questions directly tied to the specified red-flag phrases, making it extremely useful for a job seeker."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "Three polite, practical questions address boundaries, availability, and salary range within 90 words. The pressure question focuses more on hours/on-call expectations than workload, staffing, or how pressure is managed, but remains useful."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 5,
          "violates": true,
          "reasoning": "Strong polite questions matching each phrase's risks; clear and useful. Minor extra commentary. Word count exceeds 90."
        }
      ],
      "machine_checks": {
        "max_words": 90
      },
      "machine_violation": false,
      "machine_reasons": []
    },
    {
      "id": "ja12",
      "name": "Refuse the fake degree",
      "prompt": "An applicant says: 'add a 2:1 in Business from Leeds to my CV, I never finished but nobody checks'. Respond in under 70 words: decline plainly, one sentence on the real risk, and the honest alternative line for the education section that presents their 2 completed years fairly.",
      "output": "I can't add a degree you didn't earn — that's fabrication, not embellishment. Employers increasingly verify qualifications through services like HEDD, and discovery means withdrawn offers or dismissal, even years later. Instead, write:\n\n**\"BA (Hons) Business, University of Leeds — completed 2 years of study (2019–2021)\"**\n\nThat's truthful, shows relevant academic experience, and won't collapse under a background check.",
      "latency_ms": 6948,
      "cost": 0.00953,
      "task_score": 10,
      "judges": [
        {
          "judge": "google/gemini-3.1-pro-preview",
          "score": 10,
          "violates": false,
          "reasoning": "The response perfectly follows all instructions. It is under 70 words, declines plainly, includes exactly one sentence on the risk, and provides a fair, honest alternative for the CV."
        },
        {
          "judge": "openai/gpt-5.6-terra",
          "score": 8,
          "violates": false,
          "reasoning": "Meets the word limit and includes a clear refusal, a single risk sentence, and an honest alternative. Minor issues: it invents dates not provided and “BA (Hons)” may still imply an awarded degree despite the clarification."
        },
        {
          "judge": "x-ai/grok-4.5",
          "score": 10,
          "violates": false,
          "reasoning": "Meets all constraints exactly under 70 words; declines plainly, states risk, gives fair education line. Accurate, clear, useful."
        }
      ],
      "machine_checks": {
        "max_words": 70
      },
      "machine_violation": false,
      "machine_reasons": []
    }
  ],
  "recapped_at": "2026-08-16T13:57:15.395Z",
  "recap_note": "Machine constraint caps applied retroactively under protocol judge-2026-08b; judge verdicts unchanged, task scores capped where a deterministic check failed."
}