{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HE3YORGD7XMPVTFQ7R3EZAYTM3","short_pith_number":"pith:HE3YORGD","schema_version":"1.0","canonical_sha256":"39378744c3fdd8faccb0fc764c831366cf3c8fa60dc2640d7f432517d7e6b487","source":{"kind":"arxiv","id":"2407.03545","version":2},"attestation_state":"computed","paper":{"title":"On Evaluating Explanation Utility for Human-AI Decision Making in NLP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Alexander Gill, Ana Marasovi\\'c, Atreya Ghosal, Fateme Hashemi Chaleshtori, Purbid Bambroo","submitted_at":"2024-07-03T23:53:27Z","abstract_excerpt":"Is explainability a false promise? This debate has emerged from the insufficient evidence that explanations help people in situations they are introduced for. More human-centered, application-grounded evaluations of explanations are needed to settle this. Yet, with no established guidelines for such studies in NLP, researchers accustomed to standardized proxy evaluations must discover appropriate measurements, tasks, datasets, and sensible models for human-AI teams in their studies.\n  To aid with this, we first review existing metrics suitable for application-grounded evaluation. We then estab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03545","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-03T23:53:27Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"307e5e2b4b25a371683964c0d54d89cacd44bdc90cd27df1376eaf42d5643a7a","abstract_canon_sha256":"ea508c6940d320442131f9025e0bdba37fe9d2bb3c960da706273ea740b3e10d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:05.101743Z","signature_b64":"NmKohoZ2RPO2PDvoa2hZ3t+asgkqlTHKUqA+5XNwvDxuaI2hhq9MoLqZt/pjgcUkbTu8CKi3fd+iMaNuAI69AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39378744c3fdd8faccb0fc764c831366cf3c8fa60dc2640d7f432517d7e6b487","last_reissued_at":"2026-07-05T09:31:05.101280Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:05.101280Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Evaluating Explanation Utility for Human-AI Decision Making in NLP","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Alexander Gill, Ana Marasovi\\'c, Atreya Ghosal, Fateme Hashemi Chaleshtori, Purbid Bambroo","submitted_at":"2024-07-03T23:53:27Z","abstract_excerpt":"Is explainability a false promise? This debate has emerged from the insufficient evidence that explanations help people in situations they are introduced for. More human-centered, application-grounded evaluations of explanations are needed to settle this. Yet, with no established guidelines for such studies in NLP, researchers accustomed to standardized proxy evaluations must discover appropriate measurements, tasks, datasets, and sensible models for human-AI teams in their studies.\n  To aid with this, we first review existing metrics suitable for application-grounded evaluation. We then estab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03545","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03545/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03545","created_at":"2026-07-05T09:31:05.101340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03545v2","created_at":"2026-07-05T09:31:05.101340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03545","created_at":"2026-07-05T09:31:05.101340+00:00"},{"alias_kind":"pith_short_12","alias_value":"HE3YORGD7XMP","created_at":"2026-07-05T09:31:05.101340+00:00"},{"alias_kind":"pith_short_16","alias_value":"HE3YORGD7XMPVTFQ","created_at":"2026-07-05T09:31:05.101340+00:00"},{"alias_kind":"pith_short_8","alias_value":"HE3YORGD","created_at":"2026-07-05T09:31:05.101340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20293","citing_title":"Enhancing the Comprehensibility of Text Explanations via Unsupervised Concept Discovery","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3","json":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3.json","graph_json":"https://pith.science/api/pith-number/HE3YORGD7XMPVTFQ7R3EZAYTM3/graph.json","events_json":"https://pith.science/api/pith-number/HE3YORGD7XMPVTFQ7R3EZAYTM3/events.json","paper":"https://pith.science/paper/HE3YORGD"},"agent_actions":{"view_html":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3","download_json":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3.json","view_paper":"https://pith.science/paper/HE3YORGD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03545&json=true","fetch_graph":"https://pith.science/api/pith-number/HE3YORGD7XMPVTFQ7R3EZAYTM3/graph.json","fetch_events":"https://pith.science/api/pith-number/HE3YORGD7XMPVTFQ7R3EZAYTM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3/action/storage_attestation","attest_author":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3/action/author_attestation","sign_citation":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3/action/citation_signature","submit_replication":"https://pith.science/pith/HE3YORGD7XMPVTFQ7R3EZAYTM3/action/replication_record"}},"created_at":"2026-07-05T09:31:05.101340+00:00","updated_at":"2026-07-05T09:31:05.101340+00:00"}