{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QSZXCGNDJZZWJBBSVKV3HXFQLK","short_pith_number":"pith:QSZXCGND","schema_version":"1.0","canonical_sha256":"84b37119a34e73648432aaabb3dcb05a875bb50c2cf9939df1b2a08115a81f90","source":{"kind":"arxiv","id":"2410.02694","version":3},"attestation_state":"computed","paper":{"title":"HELMET: How to Evaluate Long-Context Language Models Effectively and Thoroughly","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Fleischer, Danqi Chen, Howard Yen, Ke Ding, Minmin Hou, Moshe Wasserblat, Peter Izsak, Tianyu Gao","submitted_at":"2024-10-03T17:20:11Z","abstract_excerpt":"Many benchmarks exist for evaluating long-context language models (LCLMs), yet developers often rely on synthetic tasks such as needle-in-a-haystack (NIAH) or an arbitrary subset of tasks. However, it remains unclear whether these benchmarks reflect the diverse downstream applications of LCLMs, and such inconsistencies further complicate model comparison. We investigate the underlying reasons behind these practices and find that existing benchmarks often provide noisy signals due to limited coverage of applications, insufficient context lengths, unreliable metrics, and incompatibility with bas"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02694","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T17:20:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e36d72a2827564b5e586c6ca9b529f8c2a21911decba277713ab71fb238df2f4","abstract_canon_sha256":"dbbb3fb5da8de9f64e4ff6640fdd4210f315565520886d8bc010af6f8d7260f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:11.139972Z","signature_b64":"mSzgiLrp1iUB4P4k9ByfKH2TCTw0dVb0lZKf25tYDqyU81n6cllEXhDxRs4J1n8WJ7MA6VcVb5x7NaQQxMENAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84b37119a34e73648432aaabb3dcb05a875bb50c2cf9939df1b2a08115a81f90","last_reissued_at":"2026-07-05T10:25:11.139360Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:11.139360Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HELMET: How to Evaluate Long-Context Language Models Effectively and Thoroughly","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Daniel Fleischer, Danqi Chen, Howard Yen, Ke Ding, Minmin Hou, Moshe Wasserblat, Peter Izsak, Tianyu Gao","submitted_at":"2024-10-03T17:20:11Z","abstract_excerpt":"Many benchmarks exist for evaluating long-context language models (LCLMs), yet developers often rely on synthetic tasks such as needle-in-a-haystack (NIAH) or an arbitrary subset of tasks. However, it remains unclear whether these benchmarks reflect the diverse downstream applications of LCLMs, and such inconsistencies further complicate model comparison. We investigate the underlying reasons behind these practices and find that existing benchmarks often provide noisy signals due to limited coverage of applications, insufficient context lengths, unreliable metrics, and incompatibility with bas"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02694","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02694","created_at":"2026-07-05T10:25:11.139428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02694v3","created_at":"2026-07-05T10:25:11.139428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02694","created_at":"2026-07-05T10:25:11.139428+00:00"},{"alias_kind":"pith_short_12","alias_value":"QSZXCGNDJZZW","created_at":"2026-07-05T10:25:11.139428+00:00"},{"alias_kind":"pith_short_16","alias_value":"QSZXCGNDJZZWJBBS","created_at":"2026-07-05T10:25:11.139428+00:00"},{"alias_kind":"pith_short_8","alias_value":"QSZXCGND","created_at":"2026-07-05T10:25:11.139428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07740","citing_title":"Jet-Long: Efficient Long-Context Extension with Dynamic Bifocal RoPE","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08032","citing_title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","ref_index":143,"is_internal_anchor":true},{"citing_arxiv_id":"2605.14906","citing_title":"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17328","citing_title":"MemTrace: Probing What Final Accuracy Misses in Long-Term Memory","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23231","citing_title":"PERMA: Benchmarking Personalized Memory Agents via Event-Driven Preference and Realistic Task Environments","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05257","citing_title":"Evaluating Memory in LLM Agents via Incremental Multi-Turn Interactions","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02371","citing_title":"Internalized Reasoning for Long-Context Visual Document Understanding","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":246,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12493","citing_title":"LongMemEval-V2: Evaluating Long-Term Agent Memory Toward Experienced Colleagues","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27043","citing_title":"CL-bench Life: Can Language Models Learn from Real-Life Context?","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10544","citing_title":"Where Does Long-Context Supervision Actually Go? Effective-Context Exposure Balancing","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05806","citing_title":"Retrieval from Within: An Intrinsic Capability of Attention-Based Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00072","citing_title":"XekRung Technical Report","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07809","citing_title":"PolicyLong: Towards On-Policy Context Extension","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2602.15763","citing_title":"GLM-5: from Vibe Coding to Agentic Engineering","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05806","citing_title":"Retrieval from Within: An Intrinsic Capability of Attention-Based Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06111","citing_title":"AgentCE-Bench: Agent Configurable Evaluation with Scalable Horizons and Controllable Difficulty under Lightweight Environments","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20727","citing_title":"Supplement Generation Training for Enhancing Agentic Task Performance","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK","json":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK.json","graph_json":"https://pith.science/api/pith-number/QSZXCGNDJZZWJBBSVKV3HXFQLK/graph.json","events_json":"https://pith.science/api/pith-number/QSZXCGNDJZZWJBBSVKV3HXFQLK/events.json","paper":"https://pith.science/paper/QSZXCGND"},"agent_actions":{"view_html":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK","download_json":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK.json","view_paper":"https://pith.science/paper/QSZXCGND","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02694&json=true","fetch_graph":"https://pith.science/api/pith-number/QSZXCGNDJZZWJBBSVKV3HXFQLK/graph.json","fetch_events":"https://pith.science/api/pith-number/QSZXCGNDJZZWJBBSVKV3HXFQLK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK/action/storage_attestation","attest_author":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK/action/author_attestation","sign_citation":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK/action/citation_signature","submit_replication":"https://pith.science/pith/QSZXCGNDJZZWJBBSVKV3HXFQLK/action/replication_record"}},"created_at":"2026-07-05T10:25:11.139428+00:00","updated_at":"2026-07-05T10:25:11.139428+00:00"}