{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OA3KFQ4XSLFY5B4R7EM4NMQPBY","short_pith_number":"pith:OA3KFQ4X","schema_version":"1.0","canonical_sha256":"7036a2c39792cb8e8791f919c6b20f0e04c04f84c27400d84493c2cb0c3730c0","source":{"kind":"arxiv","id":"2407.07950","version":2},"attestation_state":"computed","paper":{"title":"Rel-A.I.: An Interaction-Centered Approach To Measuring Human-LM Reliance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Dan Jurafsky, Jena D. Hwang, Kaitlyn Zhou, Maarten Sap, Nouha Dziri, Xiang Ren","submitted_at":"2024-07-10T18:00:05Z","abstract_excerpt":"The ability to communicate uncertainty, risk, and limitation is crucial for the safety of large language models. However, current evaluations of these abilities rely on simple calibration, asking whether the language generated by the model matches appropriate probabilities. Instead, evaluation of this aspect of LLM communication should focus on the behaviors of their human interlocutors: how much do they rely on what the LLM says? Here we introduce an interaction-centered evaluation framework called Rel-A.I. (pronounced \"rely\"}) that measures whether humans rely on LLM generations. We use this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07950","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-10T18:00:05Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"69679de01a2de19fae6adb43a228d418d74b54a2d6f38af0ded62b75fc5a8baa","abstract_canon_sha256":"54a48a458007a723c84a0d77ab01479df24e4d3c78b86bf25d06c79da8efab84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:13.167905Z","signature_b64":"Y2vOk1T5teclXFzsuaVqoWVMltBpnaxDf9W9L8I500upjLvMe8f0lsGk+LaFmhv7NqIBKiso0yykPQZTLfYlBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7036a2c39792cb8e8791f919c6b20f0e04c04f84c27400d84493c2cb0c3730c0","last_reissued_at":"2026-07-05T09:15:13.167404Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:13.167404Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rel-A.I.: An Interaction-Centered Approach To Measuring Human-LM Reliance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CL","authors_text":"Dan Jurafsky, Jena D. Hwang, Kaitlyn Zhou, Maarten Sap, Nouha Dziri, Xiang Ren","submitted_at":"2024-07-10T18:00:05Z","abstract_excerpt":"The ability to communicate uncertainty, risk, and limitation is crucial for the safety of large language models. However, current evaluations of these abilities rely on simple calibration, asking whether the language generated by the model matches appropriate probabilities. Instead, evaluation of this aspect of LLM communication should focus on the behaviors of their human interlocutors: how much do they rely on what the LLM says? Here we introduce an interaction-centered evaluation framework called Rel-A.I. (pronounced \"rely\"}) that measures whether humans rely on LLM generations. We use this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07950","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07950","created_at":"2026-07-05T09:15:13.167464+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07950v2","created_at":"2026-07-05T09:15:13.167464+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07950","created_at":"2026-07-05T09:15:13.167464+00:00"},{"alias_kind":"pith_short_12","alias_value":"OA3KFQ4XSLFY","created_at":"2026-07-05T09:15:13.167464+00:00"},{"alias_kind":"pith_short_16","alias_value":"OA3KFQ4XSLFY5B4R","created_at":"2026-07-05T09:15:13.167464+00:00"},{"alias_kind":"pith_short_8","alias_value":"OA3KFQ4X","created_at":"2026-07-05T09:15:13.167464+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.18045","citing_title":"From tools to thieves: Measuring and understanding public perceptions of AI through crowdsourced metaphors","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22764","citing_title":"Implicit Humanization in Everyday LLM Moral Judgments","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY","json":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY.json","graph_json":"https://pith.science/api/pith-number/OA3KFQ4XSLFY5B4R7EM4NMQPBY/graph.json","events_json":"https://pith.science/api/pith-number/OA3KFQ4XSLFY5B4R7EM4NMQPBY/events.json","paper":"https://pith.science/paper/OA3KFQ4X"},"agent_actions":{"view_html":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY","download_json":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY.json","view_paper":"https://pith.science/paper/OA3KFQ4X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07950&json=true","fetch_graph":"https://pith.science/api/pith-number/OA3KFQ4XSLFY5B4R7EM4NMQPBY/graph.json","fetch_events":"https://pith.science/api/pith-number/OA3KFQ4XSLFY5B4R7EM4NMQPBY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY/action/storage_attestation","attest_author":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY/action/author_attestation","sign_citation":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY/action/citation_signature","submit_replication":"https://pith.science/pith/OA3KFQ4XSLFY5B4R7EM4NMQPBY/action/replication_record"}},"created_at":"2026-07-05T09:15:13.167464+00:00","updated_at":"2026-07-05T09:15:13.167464+00:00"}