{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:6YVIGAUVNXBA7EKRBHU3H7DVFF","short_pith_number":"pith:6YVIGAUV","schema_version":"1.0","canonical_sha256":"f62a8302956dc20f915109e9b3fc7529677c5296eada7c12b57ce215513f877b","source":{"kind":"arxiv","id":"2607.03870","version":1},"attestation_state":"computed","paper":{"title":"Evaluating LLM Uncertainty in Long-Form Generation Using Deterministic Ground Truth","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Ido Amit, Ido Galil, Ran El-Yaniv","submitted_at":"2026-07-04T13:25:18Z","abstract_excerpt":"As LLMs generate increasingly long outputs, effective uncertainty estimation must identify errors at fine-grained levels rather than discard entire responses. While such methods exist, evaluating uncertainty at any resolution (token to an entire generation) is challenging and highly sensitive to label imperfections, making zero-noise benchmarks essential; yet, long-form generation benchmarks tend to rely on fallible labels rather than deterministic ground truth. We introduce Single-answer Atomic Long-form Target (SALT), a benchmark of six procedurally generated tasks with single deterministic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.03870","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-04T13:25:18Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"6985b0e94b939078478b955fd5752ad71f1b63d78ac195e9f27711b9ae7e6581","abstract_canon_sha256":"cd34c06d9952751cc1d0cff9cd45cf918cba0dd6e360ba99d385f1c8619d1207"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:18:11.709403Z","signature_b64":"rGSnYwjIAZmrUgb2rNc4hlDkv4Cx0bVKwVaWa5qJtv1+mylgtyYif3HEqy9XMAEWmO4NM7TDTJqptM6UcGJfCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f62a8302956dc20f915109e9b3fc7529677c5296eada7c12b57ce215513f877b","last_reissued_at":"2026-07-07T02:18:11.708746Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:18:11.708746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating LLM Uncertainty in Long-Form Generation Using Deterministic Ground Truth","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Ido Amit, Ido Galil, Ran El-Yaniv","submitted_at":"2026-07-04T13:25:18Z","abstract_excerpt":"As LLMs generate increasingly long outputs, effective uncertainty estimation must identify errors at fine-grained levels rather than discard entire responses. While such methods exist, evaluating uncertainty at any resolution (token to an entire generation) is challenging and highly sensitive to label imperfections, making zero-noise benchmarks essential; yet, long-form generation benchmarks tend to rely on fallible labels rather than deterministic ground truth. We introduce Single-answer Atomic Long-form Target (SALT), a benchmark of six procedurally generated tasks with single deterministic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.03870","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.03870/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.03870","created_at":"2026-07-07T02:18:11.708849+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.03870v1","created_at":"2026-07-07T02:18:11.708849+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.03870","created_at":"2026-07-07T02:18:11.708849+00:00"},{"alias_kind":"pith_short_12","alias_value":"6YVIGAUVNXBA","created_at":"2026-07-07T02:18:11.708849+00:00"},{"alias_kind":"pith_short_16","alias_value":"6YVIGAUVNXBA7EKR","created_at":"2026-07-07T02:18:11.708849+00:00"},{"alias_kind":"pith_short_8","alias_value":"6YVIGAUV","created_at":"2026-07-07T02:18:11.708849+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF","json":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF.json","graph_json":"https://pith.science/api/pith-number/6YVIGAUVNXBA7EKRBHU3H7DVFF/graph.json","events_json":"https://pith.science/api/pith-number/6YVIGAUVNXBA7EKRBHU3H7DVFF/events.json","paper":"https://pith.science/paper/6YVIGAUV"},"agent_actions":{"view_html":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF","download_json":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF.json","view_paper":"https://pith.science/paper/6YVIGAUV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.03870&json=true","fetch_graph":"https://pith.science/api/pith-number/6YVIGAUVNXBA7EKRBHU3H7DVFF/graph.json","fetch_events":"https://pith.science/api/pith-number/6YVIGAUVNXBA7EKRBHU3H7DVFF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF/action/storage_attestation","attest_author":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF/action/author_attestation","sign_citation":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF/action/citation_signature","submit_replication":"https://pith.science/pith/6YVIGAUVNXBA7EKRBHU3H7DVFF/action/replication_record"}},"created_at":"2026-07-07T02:18:11.708849+00:00","updated_at":"2026-07-07T02:18:11.708849+00:00"}