{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RZOB4RUX7RBGX67SIS5KRFQKTK","short_pith_number":"pith:RZOB4RUX","schema_version":"1.0","canonical_sha256":"8e5c1e4697fc426bfbf244baa8960a9aa6b39941f09bd7573d716c2797c2c178","source":{"kind":"arxiv","id":"2410.17676","version":1},"attestation_state":"computed","paper":{"title":"Towards a Similarity-adjusted Surprisal Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Clara Meister, Mario Giulianelli, Tiago Pimentel","submitted_at":"2024-10-23T08:49:51Z","abstract_excerpt":"Surprisal theory posits that the cognitive effort required to comprehend a word is determined by its contextual predictability, quantified as surprisal. Traditionally, surprisal theory treats words as distinct entities, overlooking any potential similarity between them. Giulianelli et al. (2023) address this limitation by introducing information value, a measure of predictability designed to account for similarities between communicative units. Our work leverages Ricotta and Szeidl's (2006) diversity index to extend surprisal into a metric that we term similarity-adjusted surprisal, exposing a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.17676","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-23T08:49:51Z","cross_cats_sorted":[],"title_canon_sha256":"c0783b7ade84d8cd084b00c2d2b34caf26cc1a3cd7722723ec9955c2ad48b728","abstract_canon_sha256":"93a3c0fc0a65a104f6fe8e4f15af28349b7e38d715b47944d53167be5487d14d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:42.622105Z","signature_b64":"XM6IWp6XkQLH1JxaI9Fh24hn7FMsutVEYcYCRn1UNI+87wETRZ3YI678tbnDJ/94em0aFcCbt3O1pLIzX/4zBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8e5c1e4697fc426bfbf244baa8960a9aa6b39941f09bd7573d716c2797c2c178","last_reissued_at":"2026-07-05T09:24:42.621542Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:42.621542Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards a Similarity-adjusted Surprisal Theory","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Clara Meister, Mario Giulianelli, Tiago Pimentel","submitted_at":"2024-10-23T08:49:51Z","abstract_excerpt":"Surprisal theory posits that the cognitive effort required to comprehend a word is determined by its contextual predictability, quantified as surprisal. Traditionally, surprisal theory treats words as distinct entities, overlooking any potential similarity between them. Giulianelli et al. (2023) address this limitation by introducing information value, a measure of predictability designed to account for similarities between communicative units. Our work leverages Ricotta and Szeidl's (2006) diversity index to extend surprisal into a metric that we term similarity-adjusted surprisal, exposing a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.17676","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.17676/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.17676","created_at":"2026-07-05T09:24:42.621603+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.17676v1","created_at":"2026-07-05T09:24:42.621603+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.17676","created_at":"2026-07-05T09:24:42.621603+00:00"},{"alias_kind":"pith_short_12","alias_value":"RZOB4RUX7RBG","created_at":"2026-07-05T09:24:42.621603+00:00"},{"alias_kind":"pith_short_16","alias_value":"RZOB4RUX7RBGX67S","created_at":"2026-07-05T09:24:42.621603+00:00"},{"alias_kind":"pith_short_8","alias_value":"RZOB4RUX","created_at":"2026-07-05T09:24:42.621603+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.09007","citing_title":"Vendi Information Gain: An Alternative To Mutual Information For Science And Machine Learning","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK","json":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK.json","graph_json":"https://pith.science/api/pith-number/RZOB4RUX7RBGX67SIS5KRFQKTK/graph.json","events_json":"https://pith.science/api/pith-number/RZOB4RUX7RBGX67SIS5KRFQKTK/events.json","paper":"https://pith.science/paper/RZOB4RUX"},"agent_actions":{"view_html":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK","download_json":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK.json","view_paper":"https://pith.science/paper/RZOB4RUX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.17676&json=true","fetch_graph":"https://pith.science/api/pith-number/RZOB4RUX7RBGX67SIS5KRFQKTK/graph.json","fetch_events":"https://pith.science/api/pith-number/RZOB4RUX7RBGX67SIS5KRFQKTK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK/action/storage_attestation","attest_author":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK/action/author_attestation","sign_citation":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK/action/citation_signature","submit_replication":"https://pith.science/pith/RZOB4RUX7RBGX67SIS5KRFQKTK/action/replication_record"}},"created_at":"2026-07-05T09:24:42.621603+00:00","updated_at":"2026-07-05T09:24:42.621603+00:00"}