{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:67SOKW7ITMFI6S673KFY7WWS76","short_pith_number":"pith:67SOKW7I","schema_version":"1.0","canonical_sha256":"f7e4e55be89b0a8f4bdfda8b8fdad2ff8fb589dab1ac45c0b9a481be97cfece4","source":{"kind":"arxiv","id":"2412.07961","version":1},"attestation_state":"computed","paper":{"title":"Forking Paths in Neural Text Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ari Holtzman, Eric Bigelow, Hidenori Tanaka, Tomer Ullman","submitted_at":"2024-12-10T22:57:57Z","abstract_excerpt":"Estimating uncertainty in Large Language Models (LLMs) is important for properly evaluating LLMs, and ensuring safety for users. However, prior approaches to uncertainty estimation focus on the final answer in generated text, ignoring intermediate steps that might dramatically impact the outcome. We hypothesize that there exist key forking tokens, such that re-sampling the system at those specific tokens, but not others, leads to very different outcomes. To test this empirically, we develop a novel approach to representing uncertainty dynamics across individual tokens of text generation, and a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.07961","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-10T22:57:57Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9fe869605ae31b2a054dd53720e6cfc8d89c3d93dda0db8dde8f0c1ffd9bf62b","abstract_canon_sha256":"7328f08ab847eb4c1f897ca6f43dea226735cef39497bc166cea633e2dbc59f9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:41.743422Z","signature_b64":"h88Ou59wu8/2wd+HOzZtPpoDu1G2D3tFJ3XQN/LaJx4MIT5tYVOozmH79+nNt5dcgN0TIehPDWTN92XfMJbhDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7e4e55be89b0a8f4bdfda8b8fdad2ff8fb589dab1ac45c0b9a481be97cfece4","last_reissued_at":"2026-07-05T09:47:41.742966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:41.742966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Forking Paths in Neural Text Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ari Holtzman, Eric Bigelow, Hidenori Tanaka, Tomer Ullman","submitted_at":"2024-12-10T22:57:57Z","abstract_excerpt":"Estimating uncertainty in Large Language Models (LLMs) is important for properly evaluating LLMs, and ensuring safety for users. However, prior approaches to uncertainty estimation focus on the final answer in generated text, ignoring intermediate steps that might dramatically impact the outcome. We hypothesize that there exist key forking tokens, such that re-sampling the system at those specific tokens, but not others, leads to very different outcomes. To test this empirically, we develop a novel approach to representing uncertainty dynamics across individual tokens of text generation, and a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.07961","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.07961/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.07961","created_at":"2026-07-05T09:47:41.743022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.07961v1","created_at":"2026-07-05T09:47:41.743022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.07961","created_at":"2026-07-05T09:47:41.743022+00:00"},{"alias_kind":"pith_short_12","alias_value":"67SOKW7ITMFI","created_at":"2026-07-05T09:47:41.743022+00:00"},{"alias_kind":"pith_short_16","alias_value":"67SOKW7ITMFI6S67","created_at":"2026-07-05T09:47:41.743022+00:00"},{"alias_kind":"pith_short_8","alias_value":"67SOKW7I","created_at":"2026-07-05T09:47:41.743022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05184","citing_title":"Rethinking On-Policy Self-Distillation for Thinking Models","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2510.27484","citing_title":"Thought Branches: Interpreting LLM Reasoning Requires Resampling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12112","citing_title":"When Policy Entropy Constraint Fails: Preserving Diversity in Flow-based RLHF via Perceptual Entropy","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76","json":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76.json","graph_json":"https://pith.science/api/pith-number/67SOKW7ITMFI6S673KFY7WWS76/graph.json","events_json":"https://pith.science/api/pith-number/67SOKW7ITMFI6S673KFY7WWS76/events.json","paper":"https://pith.science/paper/67SOKW7I"},"agent_actions":{"view_html":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76","download_json":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76.json","view_paper":"https://pith.science/paper/67SOKW7I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.07961&json=true","fetch_graph":"https://pith.science/api/pith-number/67SOKW7ITMFI6S673KFY7WWS76/graph.json","fetch_events":"https://pith.science/api/pith-number/67SOKW7ITMFI6S673KFY7WWS76/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76/action/timestamp_anchor","attest_storage":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76/action/storage_attestation","attest_author":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76/action/author_attestation","sign_citation":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76/action/citation_signature","submit_replication":"https://pith.science/pith/67SOKW7ITMFI6S673KFY7WWS76/action/replication_record"}},"created_at":"2026-07-05T09:47:41.743022+00:00","updated_at":"2026-07-05T09:47:41.743022+00:00"}