{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZHMUJYBKNEXUTVW4RNXIDTJL2U","short_pith_number":"pith:ZHMUJYBK","schema_version":"1.0","canonical_sha256":"c9d944e02a692f49d6dc8b6e81cd2bd51ca4b1c4665fc02a7e2cb7ebb12c11b6","source":{"kind":"arxiv","id":"2507.22209","version":1},"attestation_state":"computed","paper":{"title":"How Well Does First-Token Entropy Approximate Word Entropy as a Psycholinguistic Predictor?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Byung-Doh Oh, Christian Clark, William Schuler","submitted_at":"2025-07-29T20:12:50Z","abstract_excerpt":"Contextual entropy is a psycholinguistic measure capturing the anticipated difficulty of processing a word just before it is encountered. Recent studies have tested for entropy-related effects as a potential complement to well-known effects from surprisal. For convenience, entropy is typically estimated based on a language model's probability distribution over a word's first subword token. However, this approximation results in underestimation and potential distortion of true word entropy. To address this, we generate Monte Carlo (MC) estimates of word entropy that allow words to span a variab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22209","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-29T20:12:50Z","cross_cats_sorted":[],"title_canon_sha256":"59a8a3f99a8bae44a88bd1f99c25ad7a11dcce8e8003bccd73d78be865687353","abstract_canon_sha256":"c52bfcbebbbea9075c1995e5e5419e787a838bd6926132133e111168e7052d3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:45:18.524727Z","signature_b64":"K5FXpOyp5neUhO0N/0PBDH+7ma34pn0G1+O0urTVxGCw2d13J2nnckHLR8gqZ12ft0sK7fgdY1aJoN5DM3JmAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9d944e02a692f49d6dc8b6e81cd2bd51ca4b1c4665fc02a7e2cb7ebb12c11b6","last_reissued_at":"2026-07-05T11:45:18.524238Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:45:18.524238Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Well Does First-Token Entropy Approximate Word Entropy as a Psycholinguistic Predictor?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Byung-Doh Oh, Christian Clark, William Schuler","submitted_at":"2025-07-29T20:12:50Z","abstract_excerpt":"Contextual entropy is a psycholinguistic measure capturing the anticipated difficulty of processing a word just before it is encountered. Recent studies have tested for entropy-related effects as a potential complement to well-known effects from surprisal. For convenience, entropy is typically estimated based on a language model's probability distribution over a word's first subword token. However, this approximation results in underestimation and potential distortion of true word entropy. To address this, we generate Monte Carlo (MC) estimates of word entropy that allow words to span a variab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22209","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22209/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22209","created_at":"2026-07-05T11:45:18.524299+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22209v1","created_at":"2026-07-05T11:45:18.524299+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22209","created_at":"2026-07-05T11:45:18.524299+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHMUJYBKNEXU","created_at":"2026-07-05T11:45:18.524299+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHMUJYBKNEXUTVW4","created_at":"2026-07-05T11:45:18.524299+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHMUJYBK","created_at":"2026-07-05T11:45:18.524299+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.11922","citing_title":"LODESTAR: Trustworthy Entropy Is Navigated, Not Merely Measured -- Reinforced Polarizer Keeps a Frozen LLM from Being Confidently Misled by the Wrong Evidence","ref_index":2025,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U","json":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U.json","graph_json":"https://pith.science/api/pith-number/ZHMUJYBKNEXUTVW4RNXIDTJL2U/graph.json","events_json":"https://pith.science/api/pith-number/ZHMUJYBKNEXUTVW4RNXIDTJL2U/events.json","paper":"https://pith.science/paper/ZHMUJYBK"},"agent_actions":{"view_html":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U","download_json":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U.json","view_paper":"https://pith.science/paper/ZHMUJYBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22209&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHMUJYBKNEXUTVW4RNXIDTJL2U/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHMUJYBKNEXUTVW4RNXIDTJL2U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U/action/storage_attestation","attest_author":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U/action/author_attestation","sign_citation":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U/action/citation_signature","submit_replication":"https://pith.science/pith/ZHMUJYBKNEXUTVW4RNXIDTJL2U/action/replication_record"}},"created_at":"2026-07-05T11:45:18.524299+00:00","updated_at":"2026-07-05T11:45:18.524299+00:00"}