{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XKB45ZKZUTCRDF57FQAKOY2L2X","short_pith_number":"pith:XKB45ZKZ","schema_version":"1.0","canonical_sha256":"ba83cee559a4c51197bf2c00a7634bd5e8cd3c836d933a0b89966103e026a9fa","source":{"kind":"arxiv","id":"2502.17848","version":4},"attestation_state":"computed","paper":{"title":"LR^2Bench: Evaluating Long-chain Reflective Reasoning Capabilities of Large Language Models via Constraint Satisfaction Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiajun Zhang, Jianghao Chen, Zhenjiang Ren, Zhenlin Wei, Ziyong Li","submitted_at":"2025-02-25T04:51:17Z","abstract_excerpt":"Recent progress in Large Reasoning Models (LRMs) has significantly enhanced the reasoning abilities of Large Language Models (LLMs), empowering them to tackle increasingly complex tasks through reflection capabilities, such as making assumptions, backtracking, and self-refinement. However, effectively evaluating such reflection capabilities remains challenging due to the lack of appropriate benchmarks. To bridge this gap, we introduce LR$^2$Bench, a novel benchmark designed to evaluate the Long-chain Reflective Reasoning capabilities of LLMs. LR$^2$Bench comprises 850 samples across six Constr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.17848","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-25T04:51:17Z","cross_cats_sorted":[],"title_canon_sha256":"f17b0d6e42c04ec44fd31394832cea322cc0b32ad9b77a58b7938f10bf1ec13c","abstract_canon_sha256":"ac9e9c3705655cb80e6bded8a7dcb99321724bc4215f515dab7b3e8e74bb5f51"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:48.184917Z","signature_b64":"GRaPEFGlvBLsSKBAZwnCaVnDFiFf++jmvokPO4eHvYbLHjH93RvOsiJkylj+eoe05kzdE7R6rMs8BJ6a7DzDDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba83cee559a4c51197bf2c00a7634bd5e8cd3c836d933a0b89966103e026a9fa","last_reissued_at":"2026-07-05T11:26:48.184450Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:48.184450Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LR^2Bench: Evaluating Long-chain Reflective Reasoning Capabilities of Large Language Models via Constraint Satisfaction Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiajun Zhang, Jianghao Chen, Zhenjiang Ren, Zhenlin Wei, Ziyong Li","submitted_at":"2025-02-25T04:51:17Z","abstract_excerpt":"Recent progress in Large Reasoning Models (LRMs) has significantly enhanced the reasoning abilities of Large Language Models (LLMs), empowering them to tackle increasingly complex tasks through reflection capabilities, such as making assumptions, backtracking, and self-refinement. However, effectively evaluating such reflection capabilities remains challenging due to the lack of appropriate benchmarks. To bridge this gap, we introduce LR$^2$Bench, a novel benchmark designed to evaluate the Long-chain Reflective Reasoning capabilities of LLMs. LR$^2$Bench comprises 850 samples across six Constr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.17848","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.17848/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.17848","created_at":"2026-07-05T11:26:48.184511+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.17848v4","created_at":"2026-07-05T11:26:48.184511+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.17848","created_at":"2026-07-05T11:26:48.184511+00:00"},{"alias_kind":"pith_short_12","alias_value":"XKB45ZKZUTCR","created_at":"2026-07-05T11:26:48.184511+00:00"},{"alias_kind":"pith_short_16","alias_value":"XKB45ZKZUTCRDF57","created_at":"2026-07-05T11:26:48.184511+00:00"},{"alias_kind":"pith_short_8","alias_value":"XKB45ZKZ","created_at":"2026-07-05T11:26:48.184511+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.15180","citing_title":"PuzzleClone: A DSL-Powered Framework for Synthesizing Verifiable Data","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X","json":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X.json","graph_json":"https://pith.science/api/pith-number/XKB45ZKZUTCRDF57FQAKOY2L2X/graph.json","events_json":"https://pith.science/api/pith-number/XKB45ZKZUTCRDF57FQAKOY2L2X/events.json","paper":"https://pith.science/paper/XKB45ZKZ"},"agent_actions":{"view_html":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X","download_json":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X.json","view_paper":"https://pith.science/paper/XKB45ZKZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.17848&json=true","fetch_graph":"https://pith.science/api/pith-number/XKB45ZKZUTCRDF57FQAKOY2L2X/graph.json","fetch_events":"https://pith.science/api/pith-number/XKB45ZKZUTCRDF57FQAKOY2L2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X/action/storage_attestation","attest_author":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X/action/author_attestation","sign_citation":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X/action/citation_signature","submit_replication":"https://pith.science/pith/XKB45ZKZUTCRDF57FQAKOY2L2X/action/replication_record"}},"created_at":"2026-07-05T11:26:48.184511+00:00","updated_at":"2026-07-05T11:26:48.184511+00:00"}