{"version":"pxc-marker-validation-v1","exact":0.55,"exactCi":{"lo":0.34208200830759966,"hi":0.741804979142907},"adjacent":0.85,"adjacentCi":{"lo":0.6395767041130426,"hi":0.9476322080405041},"biasDirection":"harsh","meanDelta":-0.1,"n":20,"errors":0,"thinnest":{"key":"off-question","n":1,"exact":1,"adjacent":1,"meanDelta":0},"weakest":{"key":"waffle","n":2,"exact":0,"adjacent":0.5,"meanDelta":1.5},"cleared":false,"caveat":"Measured against a published levels-of-response rubric on hand-authored responses, not against real examiner-marked scripts. It shows whether the marker can apply a levels grid consistently; it cannot show how it would agree with a human examiner on real borderline work. Because the responses were authored to sit in a band, this is an upper bound on live agreement, not an estimate of it.","model":"llama-3.3-70b-versatile","asOf":"2026-08-03","corpusVersion":"pxc-marker-audit-set-v1","runVersion":"pxc-marker-run-v1"}