{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5NF7I72BW7TDAEZG34QOGFRVXJ","short_pith_number":"pith:5NF7I72B","schema_version":"1.0","canonical_sha256":"eb4bf47f41b7e6301326df20e31635ba7742266cd7c7dfcbb30404fb284090fa","source":{"kind":"arxiv","id":"2502.05252","version":1},"attestation_state":"computed","paper":{"title":"GSM-Infinite: How Do Your LLMs Behave over Infinitely Increasing Context Length and Reasoning Complexity?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Hongyi Liu, Yang Zhou, Yuandong Tian, Zhuoming Chen","submitted_at":"2025-02-07T17:05:25Z","abstract_excerpt":"Long-context large language models (LLMs) have recently shown strong performance in information retrieval and long-document QA. However, to tackle the most challenging intellectual problems, LLMs must reason effectively in long and complex contexts (e.g., frontier mathematical research). Studying how LLMs handle increasing reasoning complexity and context length is essential, yet existing benchmarks lack a solid basis for quantitative evaluation. Inspired by the abstraction of GSM-8K problems as computational graphs, and the ability to introduce noise by adding unnecessary nodes and edges, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05252","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-07T17:05:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7dce4a7ca79ad799f987f0cefdb4cec46803f3428827b327b48ca33d5d84003a","abstract_canon_sha256":"e1b49aacaf191c8e20720678eb9070e0d71749f1bc0ab9b38ba47b9a9b96a4d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:34.919282Z","signature_b64":"xNhHgTu+GtIbx0UExUlSvnyE101Jh54NVEsFl6uDg3STlO01UYIhqTqMYGffTruvU1sBNR0CEBEoRTjt3Fk8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb4bf47f41b7e6301326df20e31635ba7742266cd7c7dfcbb30404fb284090fa","last_reissued_at":"2026-07-05T10:11:34.918793Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:34.918793Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GSM-Infinite: How Do Your LLMs Behave over Infinitely Increasing Context Length and Reasoning Complexity?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Hongyi Liu, Yang Zhou, Yuandong Tian, Zhuoming Chen","submitted_at":"2025-02-07T17:05:25Z","abstract_excerpt":"Long-context large language models (LLMs) have recently shown strong performance in information retrieval and long-document QA. However, to tackle the most challenging intellectual problems, LLMs must reason effectively in long and complex contexts (e.g., frontier mathematical research). Studying how LLMs handle increasing reasoning complexity and context length is essential, yet existing benchmarks lack a solid basis for quantitative evaluation. Inspired by the abstraction of GSM-8K problems as computational graphs, and the ability to introduce noise by adding unnecessary nodes and edges, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05252","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05252/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05252","created_at":"2026-07-05T10:11:34.918852+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05252v1","created_at":"2026-07-05T10:11:34.918852+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05252","created_at":"2026-07-05T10:11:34.918852+00:00"},{"alias_kind":"pith_short_12","alias_value":"5NF7I72BW7TD","created_at":"2026-07-05T10:11:34.918852+00:00"},{"alias_kind":"pith_short_16","alias_value":"5NF7I72BW7TDAEZG","created_at":"2026-07-05T10:11:34.918852+00:00"},{"alias_kind":"pith_short_8","alias_value":"5NF7I72B","created_at":"2026-07-05T10:11:34.918852+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19613","citing_title":"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28079","citing_title":"ATLAS: All-round Testing of Long-context Abilities across Scales","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21350","citing_title":"Factored Causal Representation Learning for Robust Reward Modeling in RLHF","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20942","citing_title":"Bridging Structure and Language: Graph-Based Visual Reasoning for Autonomous Road Understanding","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06870","citing_title":"LEAD: Breaking the No-Recovery Bottleneck in Long-Horizon Reasoning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02780","citing_title":"MiMo-V2-Flash Technical Report","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22951","citing_title":"The Power of Power Law: Asymmetry Enables Compositional Reasoning","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ","json":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ.json","graph_json":"https://pith.science/api/pith-number/5NF7I72BW7TDAEZG34QOGFRVXJ/graph.json","events_json":"https://pith.science/api/pith-number/5NF7I72BW7TDAEZG34QOGFRVXJ/events.json","paper":"https://pith.science/paper/5NF7I72B"},"agent_actions":{"view_html":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ","download_json":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ.json","view_paper":"https://pith.science/paper/5NF7I72B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05252&json=true","fetch_graph":"https://pith.science/api/pith-number/5NF7I72BW7TDAEZG34QOGFRVXJ/graph.json","fetch_events":"https://pith.science/api/pith-number/5NF7I72BW7TDAEZG34QOGFRVXJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ/action/storage_attestation","attest_author":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ/action/author_attestation","sign_citation":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ/action/citation_signature","submit_replication":"https://pith.science/pith/5NF7I72BW7TDAEZG34QOGFRVXJ/action/replication_record"}},"created_at":"2026-07-05T10:11:34.918852+00:00","updated_at":"2026-07-05T10:11:34.918852+00:00"}