{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:TUKTBY5G3IEOZ3NDJNWA4VEYTX","short_pith_number":"pith:TUKTBY5G","canonical_record":{"source":{"id":"2602.14200","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-15T15:50:02Z","cross_cats_sorted":[],"title_canon_sha256":"d5d152a924ef4c5dba9c24fcbd53c525b67523e4ea5acda03f1ded6bc639c789","abstract_canon_sha256":"766b8daaab43b285f3f98933950c79f4fac40fd66bf2ce8bdf6b73a1932f25ed"},"schema_version":"1.0"},"canonical_sha256":"9d1530e3a6da08eceda34b6c0e54989dde99fab8df89bafa0abe12cbd7c3be8e","source":{"kind":"arxiv","id":"2602.14200","version":6},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2602.14200","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"arxiv_version","alias_value":"2602.14200v6","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.14200","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"pith_short_12","alias_value":"TUKTBY5G3IEO","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"pith_short_16","alias_value":"TUKTBY5G3IEOZ3ND","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"pith_short_8","alias_value":"TUKTBY5G","created_at":"2026-07-01T01:17:47Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:TUKTBY5G3IEOZ3NDJNWA4VEYTX","target":"record","payload":{"canonical_record":{"source":{"id":"2602.14200","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-15T15:50:02Z","cross_cats_sorted":[],"title_canon_sha256":"d5d152a924ef4c5dba9c24fcbd53c525b67523e4ea5acda03f1ded6bc639c789","abstract_canon_sha256":"766b8daaab43b285f3f98933950c79f4fac40fd66bf2ce8bdf6b73a1932f25ed"},"schema_version":"1.0"},"canonical_sha256":"9d1530e3a6da08eceda34b6c0e54989dde99fab8df89bafa0abe12cbd7c3be8e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-01T01:17:47.244051Z","signature_b64":"t2cqX1QwA5F6l8n2gIO8VbbB3oTUzaJ7XNse4GuQsVXS7x84M9f6smgQkoGZycpHfm7Z3NnNXgia4L2aoaxqBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d1530e3a6da08eceda34b6c0e54989dde99fab8df89bafa0abe12cbd7c3be8e","last_reissued_at":"2026-07-01T01:17:47.243435Z","signature_status":"signed_v1","first_computed_at":"2026-07-01T01:17:47.243435Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2602.14200","source_version":6,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-01T01:17:47Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bRdVUYMavj20zoZhJr2/L06P0x0txeJEpcco3JFgxGQ3tC3cXL6G0lTwbW8JhLX7VWlivhAVbDm3ZJWUlaUVAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T05:49:53.626459Z"},"content_sha256":"351c8720c072d5bd044fc7510bd6dd0a842f4a70ac4370425af8d5fb35ff260a","schema_version":"1.0","event_id":"sha256:351c8720c072d5bd044fc7510bd6dd0a842f4a70ac4370425af8d5fb35ff260a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:TUKTBY5G3IEOZ3NDJNWA4VEYTX","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"TS-Haystack: A Multi-Task Retrieval Benchmark for Long-Context Time-Series Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Existing time-series language models lose accuracy as contexts lengthen to a full day, yet an agentic system with classifier tools recovers performance on nine of ten tasks.","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alpay Hasanli, Elgar Fleisch, Fan Wu, Kevin O'Sullivan, Kevin Riehl, Markus Kreft, Max Rosenblattl, Maxwell A. Xu, Nicolas Zumarraga, Ning Wang, Patrick Langer, Paul Schmiedmayer, Robert Jakob, Thomas Kaar, William Tennien","submitted_at":"2026-02-15T15:50:02Z","abstract_excerpt":"Time Series Language Models (TSLMs) promise reasoning over real-world temporal data, but their ability to retrieve and reason over long time-series remains largely untested. We introduce TS-Haystack, a multi-domain retrieval benchmark with ten event-grounded question-answering tasks over contexts from 100 seconds to 24 hours, spanning direct retrieval, temporal reasoning, multi-step reasoning, and contextual anomaly detection. Existing TSLMs exhibit severe long-context degradation: accuracy declines with context length, direct-tokenization models run out of memory beyond 100 seconds on high-ra"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Existing TSLMs exhibit severe long-context degradation: accuracy declines with context length, direct-tokenization models run out of memory beyond 100 seconds on high-rate signals, and time-interval-grounded tasks collapse toward near-zero accuracy when increasing the time-series lengths... An agentic retrieval framework using specialized time-series classifier tools matches or outperforms SoTA TSLMs on 9 of 10 tasks.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The ten event-grounded tasks in TS-Haystack are representative of the core challenges in real-world long-context time-series reasoning and that observed performance drops are driven primarily by context length rather than task design, data characteristics, or model training details.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"TS-Haystack benchmark shows time-series language models degrade sharply on long contexts while an agentic retrieval system using classifier tools matches or beats them on 9 of 10 tasks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Existing time-series language models lose accuracy as contexts lengthen to a full day, yet an agentic system with classifier tools recovers performance on nine of ten tasks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"a6e5ef4a9937d1fc707cd731f4859953520b78015aeae0d63ba0134c5d6ed98b"},"source":{"id":"2602.14200","kind":"arxiv","version":6},"verdict":{"id":"3275880f-fac8-4b16-aaff-45aa5f127e11","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-15T21:39:59.646609Z","strongest_claim":"Existing TSLMs exhibit severe long-context degradation: accuracy declines with context length, direct-tokenization models run out of memory beyond 100 seconds on high-rate signals, and time-interval-grounded tasks collapse toward near-zero accuracy when increasing the time-series lengths... An agentic retrieval framework using specialized time-series classifier tools matches or outperforms SoTA TSLMs on 9 of 10 tasks.","one_line_summary":"TS-Haystack benchmark shows time-series language models degrade sharply on long contexts while an agentic retrieval system using classifier tools matches or beats them on 9 of 10 tasks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The ten event-grounded tasks in TS-Haystack are representative of the core challenges in real-world long-context time-series reasoning and that observed performance drops are driven primarily by context length rather than task design, data characteristics, or model training details.","pith_extraction_headline":"Existing time-series language models lose accuracy as contexts lengthen to a full day, yet an agentic system with classifier tools recovers performance on nine of ten tasks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.14200/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"3275880f-fac8-4b16-aaff-45aa5f127e11"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-01T01:17:47Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IFtc66u7JbsW0f7JSk6v35/f86SuMZPVGB6qrM8YaNVGn/8jPPOyc21C5nsHe9wmdh6XvB+At3qhFv50VokzDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T05:49:53.627326Z"},"content_sha256":"fb09df5b7317bdaff090b5bc15474e7ce17c7d1ee18ab8db4191a57c9546570e","schema_version":"1.0","event_id":"sha256:fb09df5b7317bdaff090b5bc15474e7ce17c7d1ee18ab8db4191a57c9546570e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX/bundle.json","state_url":"https://pith.science/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T05:49:53Z","links":{"resolver":"https://pith.science/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX","bundle":"https://pith.science/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX/bundle.json","state":"https://pith.science/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX/state.json","well_known_bundle":"https://pith.science/.well-known/pith/TUKTBY5G3IEOZ3NDJNWA4VEYTX/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:TUKTBY5G3IEOZ3NDJNWA4VEYTX","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"766b8daaab43b285f3f98933950c79f4fac40fd66bf2ce8bdf6b73a1932f25ed","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-15T15:50:02Z","title_canon_sha256":"d5d152a924ef4c5dba9c24fcbd53c525b67523e4ea5acda03f1ded6bc639c789"},"schema_version":"1.0","source":{"id":"2602.14200","kind":"arxiv","version":6}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2602.14200","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"arxiv_version","alias_value":"2602.14200v6","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.14200","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"pith_short_12","alias_value":"TUKTBY5G3IEO","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"pith_short_16","alias_value":"TUKTBY5G3IEOZ3ND","created_at":"2026-07-01T01:17:47Z"},{"alias_kind":"pith_short_8","alias_value":"TUKTBY5G","created_at":"2026-07-01T01:17:47Z"}],"graph_snapshots":[{"event_id":"sha256:fb09df5b7317bdaff090b5bc15474e7ce17c7d1ee18ab8db4191a57c9546570e","target":"graph","created_at":"2026-07-01T01:17:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Existing TSLMs exhibit severe long-context degradation: accuracy declines with context length, direct-tokenization models run out of memory beyond 100 seconds on high-rate signals, and time-interval-grounded tasks collapse toward near-zero accuracy when increasing the time-series lengths... An agentic retrieval framework using specialized time-series classifier tools matches or outperforms SoTA TSLMs on 9 of 10 tasks."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The ten event-grounded tasks in TS-Haystack are representative of the core challenges in real-world long-context time-series reasoning and that observed performance drops are driven primarily by context length rather than task design, data characteristics, or model training details."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"TS-Haystack benchmark shows time-series language models degrade sharply on long contexts while an agentic retrieval system using classifier tools matches or beats them on 9 of 10 tasks."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Existing time-series language models lose accuracy as contexts lengthen to a full day, yet an agentic system with classifier tools recovers performance on nine of ten tasks."}],"snapshot_sha256":"a6e5ef4a9937d1fc707cd731f4859953520b78015aeae0d63ba0134c5d6ed98b"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2602.14200/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Time Series Language Models (TSLMs) promise reasoning over real-world temporal data, but their ability to retrieve and reason over long time-series remains largely untested. We introduce TS-Haystack, a multi-domain retrieval benchmark with ten event-grounded question-answering tasks over contexts from 100 seconds to 24 hours, spanning direct retrieval, temporal reasoning, multi-step reasoning, and contextual anomaly detection. Existing TSLMs exhibit severe long-context degradation: accuracy declines with context length, direct-tokenization models run out of memory beyond 100 seconds on high-ra","authors_text":"Alpay Hasanli, Elgar Fleisch, Fan Wu, Kevin O'Sullivan, Kevin Riehl, Markus Kreft, Max Rosenblattl, Maxwell A. Xu, Nicolas Zumarraga, Ning Wang, Patrick Langer, Paul Schmiedmayer, Robert Jakob, Thomas Kaar, William Tennien","cross_cats":[],"headline":"Existing time-series language models lose accuracy as contexts lengthen to a full day, yet an agentic system with classifier tools recovers performance on nine of ten tasks.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-15T15:50:02Z","title":"TS-Haystack: A Multi-Task Retrieval Benchmark for Long-Context Time-Series Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.14200","kind":"arxiv","version":6},"verdict":{"created_at":"2026-05-15T21:39:59.646609Z","id":"3275880f-fac8-4b16-aaff-45aa5f127e11","model_set":{"reader":"grok-4.3"},"one_line_summary":"TS-Haystack benchmark shows time-series language models degrade sharply on long contexts while an agentic retrieval system using classifier tools matches or beats them on 9 of 10 tasks.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Existing time-series language models lose accuracy as contexts lengthen to a full day, yet an agentic system with classifier tools recovers performance on nine of ten tasks.","strongest_claim":"Existing TSLMs exhibit severe long-context degradation: accuracy declines with context length, direct-tokenization models run out of memory beyond 100 seconds on high-rate signals, and time-interval-grounded tasks collapse toward near-zero accuracy when increasing the time-series lengths... An agentic retrieval framework using specialized time-series classifier tools matches or outperforms SoTA TSLMs on 9 of 10 tasks.","weakest_assumption":"The ten event-grounded tasks in TS-Haystack are representative of the core challenges in real-world long-context time-series reasoning and that observed performance drops are driven primarily by context length rather than task design, data characteristics, or model training details."}},"verdict_id":"3275880f-fac8-4b16-aaff-45aa5f127e11"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:351c8720c072d5bd044fc7510bd6dd0a842f4a70ac4370425af8d5fb35ff260a","target":"record","created_at":"2026-07-01T01:17:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"766b8daaab43b285f3f98933950c79f4fac40fd66bf2ce8bdf6b73a1932f25ed","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-15T15:50:02Z","title_canon_sha256":"d5d152a924ef4c5dba9c24fcbd53c525b67523e4ea5acda03f1ded6bc639c789"},"schema_version":"1.0","source":{"id":"2602.14200","kind":"arxiv","version":6}},"canonical_sha256":"9d1530e3a6da08eceda34b6c0e54989dde99fab8df89bafa0abe12cbd7c3be8e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9d1530e3a6da08eceda34b6c0e54989dde99fab8df89bafa0abe12cbd7c3be8e","first_computed_at":"2026-07-01T01:17:47.243435Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-01T01:17:47.243435Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"t2cqX1QwA5F6l8n2gIO8VbbB3oTUzaJ7XNse4GuQsVXS7x84M9f6smgQkoGZycpHfm7Z3NnNXgia4L2aoaxqBA==","signature_status":"signed_v1","signed_at":"2026-07-01T01:17:47.244051Z","signed_message":"canonical_sha256_bytes"},"source_id":"2602.14200","source_kind":"arxiv","source_version":6}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:351c8720c072d5bd044fc7510bd6dd0a842f4a70ac4370425af8d5fb35ff260a","sha256:fb09df5b7317bdaff090b5bc15474e7ce17c7d1ee18ab8db4191a57c9546570e"],"state_sha256":"8e3b979e5205a920d512a9f102b06575c9c7d362954dd6198bc20aa2df51f737"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fBWqYGn7pZJLRwXz1c4zg4g15P8YGIT+5zKozmMWzFgASK4E1Gb7QCBkzEBME75e/jwFxS0hHY0F/MF0MdtFBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T05:49:53.631656Z","bundle_sha256":"edea30dcebfbd40fae9c276d8f2e4e3df6dd61f76dc5f8921d9c12386c963942"}}