{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V4UNGYPWIRHBXZZSTEELFCNI36","short_pith_number":"pith:V4UNGYPW","schema_version":"1.0","canonical_sha256":"af28d361f6444e1be7329908b289a8dfa5943443fc0522c07534f42687bb895b","source":{"kind":"arxiv","id":"2406.16008","version":2},"attestation_state":"computed","paper":{"title":"Found in the Middle: Calibrating Positional Attention Bias Improves Long Context Utilization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhishek Kumar, Alexander Ratner, Cheng-Yu Hsieh, Chen-Yu Lee, Chun-Liang Li, James Glass, Long T. Le, Ranjay Krishna, Tomas Pfister, Yung-Sung Chuang, Zifeng Wang","submitted_at":"2024-06-23T04:35:42Z","abstract_excerpt":"Large language models (LLMs), even when specifically trained to process long input contexts, struggle to capture relevant information located in the middle of their input. This phenomenon has been known as the lost-in-the-middle problem. In this work, we make three contributions. First, we set out to understand the factors that cause this phenomenon. In doing so, we establish a connection between lost-in-the-middle to LLMs' intrinsic attention bias: LLMs exhibit a U-shaped attention bias where the tokens at the beginning and at the end of its input receive higher attention, regardless of their"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16008","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-23T04:35:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3cc6d600140e2464fae71f19c154de80694a32b3735fba5694f98a3dad81a8b7","abstract_canon_sha256":"ef99ed0ea79c1b519bc70b2dadd20204213fd50217434e8633b22f2938b88783"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:35.343565Z","signature_b64":"OkJCCKJrx5J7mV/CGDf61r1pz7+4akv48tPfwp3ZwIC6qYgtmZkE7qTeE3tYP8ntvq7D5eXIspTH3+atP83+AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af28d361f6444e1be7329908b289a8dfa5943443fc0522c07534f42687bb895b","last_reissued_at":"2026-07-05T08:39:35.343087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:35.343087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Found in the Middle: Calibrating Positional Attention Bias Improves Long Context Utilization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Abhishek Kumar, Alexander Ratner, Cheng-Yu Hsieh, Chen-Yu Lee, Chun-Liang Li, James Glass, Long T. Le, Ranjay Krishna, Tomas Pfister, Yung-Sung Chuang, Zifeng Wang","submitted_at":"2024-06-23T04:35:42Z","abstract_excerpt":"Large language models (LLMs), even when specifically trained to process long input contexts, struggle to capture relevant information located in the middle of their input. This phenomenon has been known as the lost-in-the-middle problem. In this work, we make three contributions. First, we set out to understand the factors that cause this phenomenon. In doing so, we establish a connection between lost-in-the-middle to LLMs' intrinsic attention bias: LLMs exhibit a U-shaped attention bias where the tokens at the beginning and at the end of its input receive higher attention, regardless of their"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16008","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16008/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16008","created_at":"2026-07-05T08:39:35.343142+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16008v2","created_at":"2026-07-05T08:39:35.343142+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16008","created_at":"2026-07-05T08:39:35.343142+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4UNGYPWIRHB","created_at":"2026-07-05T08:39:35.343142+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4UNGYPWIRHBXZZS","created_at":"2026-07-05T08:39:35.343142+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4UNGYPW","created_at":"2026-07-05T08:39:35.343142+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27705","citing_title":"Mitigating Position Bias in Transformers via Layer-Specific Positional Embedding Scaling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22102","citing_title":"Mitigating Coordinate Prediction Bias from Positional Encoding Failures","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14197","citing_title":"The PICCO Framework for Large Language Model Prompting: A Taxonomy and Reference Architecture for Prompt Structure","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36","json":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36.json","graph_json":"https://pith.science/api/pith-number/V4UNGYPWIRHBXZZSTEELFCNI36/graph.json","events_json":"https://pith.science/api/pith-number/V4UNGYPWIRHBXZZSTEELFCNI36/events.json","paper":"https://pith.science/paper/V4UNGYPW"},"agent_actions":{"view_html":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36","download_json":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36.json","view_paper":"https://pith.science/paper/V4UNGYPW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16008&json=true","fetch_graph":"https://pith.science/api/pith-number/V4UNGYPWIRHBXZZSTEELFCNI36/graph.json","fetch_events":"https://pith.science/api/pith-number/V4UNGYPWIRHBXZZSTEELFCNI36/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36/action/storage_attestation","attest_author":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36/action/author_attestation","sign_citation":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36/action/citation_signature","submit_replication":"https://pith.science/pith/V4UNGYPWIRHBXZZSTEELFCNI36/action/replication_record"}},"created_at":"2026-07-05T08:39:35.343142+00:00","updated_at":"2026-07-05T08:39:35.343142+00:00"}