{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VS4JTPKLID5MSBOTBGUEE62CAR","short_pith_number":"pith:VS4JTPKL","schema_version":"1.0","canonical_sha256":"acb899bd4b40fac905d309a8427b420456b5a89a8093b23e0258a57ab723c0b3","source":{"kind":"arxiv","id":"2411.07130","version":3},"attestation_state":"computed","paper":{"title":"On Many-Shot In-Context Learning for Long-Context Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kaijian Zou, Lu Wang, Muhammad Khalifa","submitted_at":"2024-11-11T17:00:59Z","abstract_excerpt":"Many-shot in-context learning (ICL) has emerged as a unique setup to both utilize and test the ability of large language models to handle long context. This paper delves into long-context language model (LCLM) evaluation through many-shot ICL. We first ask: what types of ICL tasks benefit from additional demonstrations, and how effective are they in evaluating LCLMs? We find that classification and summarization tasks show performance improvements with additional demonstrations, while translation and reasoning tasks do not exhibit clear trends. Next, we investigate the extent to which differen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.07130","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-11T17:00:59Z","cross_cats_sorted":[],"title_canon_sha256":"c97d64b2bf09fa9f38f99a6f48ad3aafe18eefc3cd64d8e56664eaf39f6fa48a","abstract_canon_sha256":"670f8ee01edc296a07c46b2d1d8e2cc1d35474fbbde311310f4a9404df7e9b05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:20:06.589371Z","signature_b64":"VR1Lw/xNk9suDVHwJ015Z5YH1ACP1lkBWd+KGBQKhO4UA4pdp2+u+3jtG6iYow8ZC2kDJONChmpDVBEeWBNLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"acb899bd4b40fac905d309a8427b420456b5a89a8093b23e0258a57ab723c0b3","last_reissued_at":"2026-07-05T11:20:06.588940Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:20:06.588940Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Many-Shot In-Context Learning for Long-Context Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kaijian Zou, Lu Wang, Muhammad Khalifa","submitted_at":"2024-11-11T17:00:59Z","abstract_excerpt":"Many-shot in-context learning (ICL) has emerged as a unique setup to both utilize and test the ability of large language models to handle long context. This paper delves into long-context language model (LCLM) evaluation through many-shot ICL. We first ask: what types of ICL tasks benefit from additional demonstrations, and how effective are they in evaluating LCLMs? We find that classification and summarization tasks show performance improvements with additional demonstrations, while translation and reasoning tasks do not exhibit clear trends. Next, we investigate the extent to which differen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.07130","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.07130/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.07130","created_at":"2026-07-05T11:20:06.588995+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.07130v3","created_at":"2026-07-05T11:20:06.588995+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.07130","created_at":"2026-07-05T11:20:06.588995+00:00"},{"alias_kind":"pith_short_12","alias_value":"VS4JTPKLID5M","created_at":"2026-07-05T11:20:06.588995+00:00"},{"alias_kind":"pith_short_16","alias_value":"VS4JTPKLID5MSBOT","created_at":"2026-07-05T11:20:06.588995+00:00"},{"alias_kind":"pith_short_8","alias_value":"VS4JTPKL","created_at":"2026-07-05T11:20:06.588995+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":267,"is_internal_anchor":false},{"citing_arxiv_id":"2507.20906","citing_title":"Soft Head Selection for Injecting ICL-Derived Task Embeddings","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08212","citing_title":"LLMs with in-context learning for Algorithmic Theoretical Physics","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR","json":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR.json","graph_json":"https://pith.science/api/pith-number/VS4JTPKLID5MSBOTBGUEE62CAR/graph.json","events_json":"https://pith.science/api/pith-number/VS4JTPKLID5MSBOTBGUEE62CAR/events.json","paper":"https://pith.science/paper/VS4JTPKL"},"agent_actions":{"view_html":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR","download_json":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR.json","view_paper":"https://pith.science/paper/VS4JTPKL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.07130&json=true","fetch_graph":"https://pith.science/api/pith-number/VS4JTPKLID5MSBOTBGUEE62CAR/graph.json","fetch_events":"https://pith.science/api/pith-number/VS4JTPKLID5MSBOTBGUEE62CAR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR/action/storage_attestation","attest_author":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR/action/author_attestation","sign_citation":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR/action/citation_signature","submit_replication":"https://pith.science/pith/VS4JTPKLID5MSBOTBGUEE62CAR/action/replication_record"}},"created_at":"2026-07-05T11:20:06.588995+00:00","updated_at":"2026-07-05T11:20:06.588995+00:00"}