{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V7RONU2HBWWPMJEMJ6YPCRJLDT","short_pith_number":"pith:V7RONU2H","schema_version":"1.0","canonical_sha256":"afe2e6d3470dacf6248c4fb0f1452b1cf176c73395c91eae0d8b29e9743168dd","source":{"kind":"arxiv","id":"2407.07799","version":2},"attestation_state":"computed","paper":{"title":"Attribute or Abstain: Large Language Models as Long Document Assistants","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Iryna Gurevych, Jan Buchmann, Xiao Liu","submitted_at":"2024-07-10T16:16:02Z","abstract_excerpt":"LLMs can help humans working with long documents, but are known to hallucinate. Attribution can increase trust in LLM responses: The LLM provides evidence that supports its response, which enhances verifiability. Existing approaches to attribution have only been evaluated in RAG settings, where the initial retrieval confounds LLM performance. This is crucially different from the long document setting, where retrieval is not needed, but could help. Thus, a long document specific evaluation of attribution is missing. To fill this gap, we present LAB, a benchmark of 6 diverse long document tasks "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07799","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-10T16:16:02Z","cross_cats_sorted":[],"title_canon_sha256":"3eb0f80b3b7b5d9ca6e718ebd316a31dd22509fcb2176446d8f50ba8fc258200","abstract_canon_sha256":"36cc26b3eed2d02222724b5d9d17f0ddd7c745ba119f03de9fd87f26738e5eb9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:36.601073Z","signature_b64":"l8bhU6drmZ0+LoXojH+Nryr5Ty/jLTColGqZj1eol+k7v0wFuxRKE8jFOvTbz9efsAJzPr2Jt4QOmMawwplRDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"afe2e6d3470dacf6248c4fb0f1452b1cf176c73395c91eae0d8b29e9743168dd","last_reissued_at":"2026-07-05T09:24:36.600546Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:36.600546Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attribute or Abstain: Large Language Models as Long Document Assistants","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Iryna Gurevych, Jan Buchmann, Xiao Liu","submitted_at":"2024-07-10T16:16:02Z","abstract_excerpt":"LLMs can help humans working with long documents, but are known to hallucinate. Attribution can increase trust in LLM responses: The LLM provides evidence that supports its response, which enhances verifiability. Existing approaches to attribution have only been evaluated in RAG settings, where the initial retrieval confounds LLM performance. This is crucially different from the long document setting, where retrieval is not needed, but could help. Thus, a long document specific evaluation of attribution is missing. To fill this gap, we present LAB, a benchmark of 6 diverse long document tasks "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07799","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07799/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07799","created_at":"2026-07-05T09:24:36.600606+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07799v2","created_at":"2026-07-05T09:24:36.600606+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07799","created_at":"2026-07-05T09:24:36.600606+00:00"},{"alias_kind":"pith_short_12","alias_value":"V7RONU2HBWWP","created_at":"2026-07-05T09:24:36.600606+00:00"},{"alias_kind":"pith_short_16","alias_value":"V7RONU2HBWWPMJEM","created_at":"2026-07-05T09:24:36.600606+00:00"},{"alias_kind":"pith_short_8","alias_value":"V7RONU2H","created_at":"2026-07-05T09:24:36.600606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23170","citing_title":"Positional Failures in Long-Context LLMs: A Blind Spot in Reasoning Benchmarks","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT","json":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT.json","graph_json":"https://pith.science/api/pith-number/V7RONU2HBWWPMJEMJ6YPCRJLDT/graph.json","events_json":"https://pith.science/api/pith-number/V7RONU2HBWWPMJEMJ6YPCRJLDT/events.json","paper":"https://pith.science/paper/V7RONU2H"},"agent_actions":{"view_html":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT","download_json":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT.json","view_paper":"https://pith.science/paper/V7RONU2H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07799&json=true","fetch_graph":"https://pith.science/api/pith-number/V7RONU2HBWWPMJEMJ6YPCRJLDT/graph.json","fetch_events":"https://pith.science/api/pith-number/V7RONU2HBWWPMJEMJ6YPCRJLDT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT/action/storage_attestation","attest_author":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT/action/author_attestation","sign_citation":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT/action/citation_signature","submit_replication":"https://pith.science/pith/V7RONU2HBWWPMJEMJ6YPCRJLDT/action/replication_record"}},"created_at":"2026-07-05T09:24:36.600606+00:00","updated_at":"2026-07-05T09:24:36.600606+00:00"}