{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A6W76LEDHA6GLUVLBFDTLFXUDI","short_pith_number":"pith:A6W76LED","schema_version":"1.0","canonical_sha256":"07adff2c83383c65d2ab09473596f41a162913d47d01183f82f1e5e43e7446ea","source":{"kind":"arxiv","id":"2404.02060","version":3},"attestation_state":"computed","paper":{"title":"Long-context LLMs Struggle with Long In-context Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ge Zhang, Quy Duc Do, Tianle Li, Wenhu Chen, Xiang Yue","submitted_at":"2024-04-02T15:59:11Z","abstract_excerpt":"Large Language Models (LLMs) have made significant strides in handling long sequences. Some models like Gemini could even to be capable of dealing with millions of tokens. However, their performance evaluation has largely been confined to metrics like perplexity and synthetic tasks, which may not fully capture their true abilities in more challenging, real-world scenarios. We introduce a benchmark (LongICLBench) for long in-context learning in extreme-label classification using six datasets with 28 to 174 classes and input lengths from 2K to 50K tokens. Our benchmark requires LLMs to comprehen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.02060","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-02T15:59:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"85482fa79ae20c8cd13dd9ae5535d730f83ca6b32e781149d728e8e87a515020","abstract_canon_sha256":"1dd80fdc48c1bea109c1a377be9c4f91fdc82188e97dbfdc17db7398b71a9207"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:32.781211Z","signature_b64":"ClBaHeMbkcaLf5ytEXk3Rf27Ch2nfyTntChTVH8Zp3dvBbaJKLvaX3Lj0SNPAMXYqwnL6hDs0bG20M7y7PZnBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07adff2c83383c65d2ab09473596f41a162913d47d01183f82f1e5e43e7446ea","last_reissued_at":"2026-07-05T08:30:32.780663Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:32.780663Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long-context LLMs Struggle with Long In-context Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ge Zhang, Quy Duc Do, Tianle Li, Wenhu Chen, Xiang Yue","submitted_at":"2024-04-02T15:59:11Z","abstract_excerpt":"Large Language Models (LLMs) have made significant strides in handling long sequences. Some models like Gemini could even to be capable of dealing with millions of tokens. However, their performance evaluation has largely been confined to metrics like perplexity and synthetic tasks, which may not fully capture their true abilities in more challenging, real-world scenarios. We introduce a benchmark (LongICLBench) for long in-context learning in extreme-label classification using six datasets with 28 to 174 classes and input lengths from 2K to 50K tokens. Our benchmark requires LLMs to comprehen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.02060","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.02060/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.02060","created_at":"2026-07-05T08:30:32.780725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.02060v3","created_at":"2026-07-05T08:30:32.780725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.02060","created_at":"2026-07-05T08:30:32.780725+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6W76LEDHA6G","created_at":"2026-07-05T08:30:32.780725+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6W76LEDHA6GLUVL","created_at":"2026-07-05T08:30:32.780725+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6W76LED","created_at":"2026-07-05T08:30:32.780725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07894","citing_title":"DD-GEPA: Prompt Optimization for Dialogue Disentanglement Focusing on Task Instruction and Utterance Representation","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06906","citing_title":"EASE-TTT: Evidence-Aligned Selective Test-Time Training for Long-Context Question Answering","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05436","citing_title":"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29742","citing_title":"MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29844","citing_title":"MATCH: Modulating Attention via In-Context Retrieval for Long-Context Transformers","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25954","citing_title":"Step-TP: A Grounded, Step-Level Dataset with Chain-of-Thought Reasoning for LLM-Guided Tensor Program Optimization","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27705","citing_title":"Mitigating Position Bias in Transformers via Layer-Specific Positional Embedding Scaling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28713","citing_title":"Thinking as Compression: Your Reasoning Model is Secretly a Context Compressor","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29512","citing_title":"MINDGAMES: A Live Arena for Evaluating Social and Strategic Reasoning in Multi-Agent LLMs","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2409.02428","citing_title":"Language Models as Efficient Reward Function Searchers for Custom-Environment Multi-Objective Reinforcement","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2504.14044","citing_title":"Multi-Stage Retrieval for Operational Technology Cybersecurity Compliance Using Large Language Models: A Railway Casestudy","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05583","citing_title":"KG-HTC: Integrating Knowledge Graphs into LLMs for Effective Zero-shot Hierarchical Text Classification","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12402","citing_title":"Automated Profile Inference with Language Model Agents","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10708","citing_title":"SafeTrans: LLM-assisted Transpilation from C to Rust","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21768","citing_title":"Memory-R2: Fair Credit Assignment for Long-Horizon Memory-Augmented LLM Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2506.11060","citing_title":"Code Researcher: Deep Research Agent for Large Systems Code and Commit History","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19259","citing_title":"ERFSL: An Efficient Reward Function Searcher via Language Models for Custom-Environment Multi-Objective Optimization (Student Abstract)","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17881","citing_title":"POPI: Personalizing LLMs via Optimized Natural Language Preference Inference","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00309","citing_title":"Retrieval-Augmented Generation with Graphs (GraphRAG)","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":250,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12703","citing_title":"MMCL-Bench: Multimodal Context Learning from Visual Rules, Procedures, and Evidence","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25924","citing_title":"Generative AI-Based Virtual Assistant using Retrieval-Augmented Generation: An evaluation study for bachelor projects","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08212","citing_title":"LLMs with in-context learning for Algorithmic Theoretical Physics","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22234","citing_title":"GR-Evolve: Design-Adaptive Global Routing via LLM-Driven Algorithm Evolution","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22427","citing_title":"Automation-Exploit: A Multi-Agent LLM Framework for Adaptive Offensive Security with Digital Twin-Based Risk-Mitigated Exploitation","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI","json":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI.json","graph_json":"https://pith.science/api/pith-number/A6W76LEDHA6GLUVLBFDTLFXUDI/graph.json","events_json":"https://pith.science/api/pith-number/A6W76LEDHA6GLUVLBFDTLFXUDI/events.json","paper":"https://pith.science/paper/A6W76LED"},"agent_actions":{"view_html":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI","download_json":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI.json","view_paper":"https://pith.science/paper/A6W76LED","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.02060&json=true","fetch_graph":"https://pith.science/api/pith-number/A6W76LEDHA6GLUVLBFDTLFXUDI/graph.json","fetch_events":"https://pith.science/api/pith-number/A6W76LEDHA6GLUVLBFDTLFXUDI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI/action/storage_attestation","attest_author":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI/action/author_attestation","sign_citation":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI/action/citation_signature","submit_replication":"https://pith.science/pith/A6W76LEDHA6GLUVLBFDTLFXUDI/action/replication_record"}},"created_at":"2026-07-05T08:30:32.780725+00:00","updated_at":"2026-07-05T08:30:32.780725+00:00"}