{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:QSZXCGNDJZZWJBBSVKV3HXFQLK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"dbbb3fb5da8de9f64e4ff6640fdd4210f315565520886d8bc010af6f8d7260f0","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T17:20:11Z","title_canon_sha256":"e36d72a2827564b5e586c6ca9b529f8c2a21911decba277713ab71fb238df2f4"},"schema_version":"1.0","source":{"id":"2410.02694","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.02694","created_at":"2026-07-05T10:25:11Z"},{"alias_kind":"arxiv_version","alias_value":"2410.02694v3","created_at":"2026-07-05T10:25:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02694","created_at":"2026-07-05T10:25:11Z"},{"alias_kind":"pith_short_12","alias_value":"QSZXCGNDJZZW","created_at":"2026-07-05T10:25:11Z"},{"alias_kind":"pith_short_16","alias_value":"QSZXCGNDJZZWJBBS","created_at":"2026-07-05T10:25:11Z"},{"alias_kind":"pith_short_8","alias_value":"QSZXCGND","created_at":"2026-07-05T10:25:11Z"}],"graph_snapshots":[{"event_id":"sha256:61e80922516a416304cd701171f4819f01212ca31b0e76fe8508ac2a5a5da943","target":"graph","created_at":"2026-07-05T10:25:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.02694/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Many benchmarks exist for evaluating long-context language models (LCLMs), yet developers often rely on synthetic tasks such as needle-in-a-haystack (NIAH) or an arbitrary subset of tasks. However, it remains unclear whether these benchmarks reflect the diverse downstream applications of LCLMs, and such inconsistencies further complicate model comparison. We investigate the underlying reasons behind these practices and find that existing benchmarks often provide noisy signals due to limited coverage of applications, insufficient context lengths, unreliable metrics, and incompatibility with bas","authors_text":"Daniel Fleischer, Danqi Chen, Howard Yen, Ke Ding, Minmin Hou, Moshe Wasserblat, Peter Izsak, Tianyu Gao","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T17:20:11Z","title":"HELMET: How to Evaluate Long-Context Language Models Effectively and Thoroughly"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02694","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:eef3d6855ad441a662175474e14c9e4624396f9e049893b6274bfea0cd63fea0","target":"record","created_at":"2026-07-05T10:25:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"dbbb3fb5da8de9f64e4ff6640fdd4210f315565520886d8bc010af6f8d7260f0","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-03T17:20:11Z","title_canon_sha256":"e36d72a2827564b5e586c6ca9b529f8c2a21911decba277713ab71fb238df2f4"},"schema_version":"1.0","source":{"id":"2410.02694","kind":"arxiv","version":3}},"canonical_sha256":"84b37119a34e73648432aaabb3dcb05a875bb50c2cf9939df1b2a08115a81f90","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"84b37119a34e73648432aaabb3dcb05a875bb50c2cf9939df1b2a08115a81f90","first_computed_at":"2026-07-05T10:25:11.139360Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:25:11.139360Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"mSzgiLrp1iUB4P4k9ByfKH2TCTw0dVb0lZKf25tYDqyU81n6cllEXhDxRs4J1n8WJ7MA6VcVb5x7NaQQxMENAg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:25:11.139972Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.02694","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:eef3d6855ad441a662175474e14c9e4624396f9e049893b6274bfea0cd63fea0","sha256:61e80922516a416304cd701171f4819f01212ca31b0e76fe8508ac2a5a5da943"],"state_sha256":"b365d17b9baa8f57771aa9b48218309a1eff67bb240d01b67eaaee7c1f6a3b95"}