{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QWUJIRD23WRVONMTSGXMTICH2O","short_pith_number":"pith:QWUJIRD2","schema_version":"1.0","canonical_sha256":"85a894447adda357359391aec9a047d3889aa9eac6beb89a48d41fb4e392a3fa","source":{"kind":"arxiv","id":"2507.14295","version":2},"attestation_state":"computed","paper":{"title":"A Simple \"Try Again\" Can Elicit Multi-Turn LLM Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Avirup Sil, Chenwei Xu, Han Liu, Licheng Liu, Linjie Li, Manling Li, Yiping Lu, Zihan Wang","submitted_at":"2025-07-18T18:07:38Z","abstract_excerpt":"Multi-turn problem solving is critical yet challenging for Large Reasoning Models (LRMs) to reflect on their reasoning and revise from feedback. Existing Reinforcement Learning (RL) methods train large reasoning models on a single-turn paradigm with verifiable rewards. However, we observe that models trained with existing RL paradigms often lose their ability to solve problems across multiple turns and struggle to revise answers based on contextual feedback, leading to repetitive responses. We ask: can LRMs learn to reflect their answers in a multi-turn context? In this work, we find that trai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.14295","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-18T18:07:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"48980a58a4e7d53b08bea136bfefe49c8b8630cfa54cbeb8521609689e9aeb11","abstract_canon_sha256":"e4b4740921abf455aeda7194deb2c6e3ea5c50ce98b9437a822f101ce94995d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:57:38.048044Z","signature_b64":"RwX7pKQ1mq9QkHt4UGnp7RIvSVDr7kwYbId8pWG3nZbgFqKdrqdiW4UdG/AUaPOK0geOHmPWwMFyqhSJKTJ+Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85a894447adda357359391aec9a047d3889aa9eac6beb89a48d41fb4e392a3fa","last_reissued_at":"2026-07-05T11:57:38.047577Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:57:38.047577Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Simple \"Try Again\" Can Elicit Multi-Turn LLM Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Avirup Sil, Chenwei Xu, Han Liu, Licheng Liu, Linjie Li, Manling Li, Yiping Lu, Zihan Wang","submitted_at":"2025-07-18T18:07:38Z","abstract_excerpt":"Multi-turn problem solving is critical yet challenging for Large Reasoning Models (LRMs) to reflect on their reasoning and revise from feedback. Existing Reinforcement Learning (RL) methods train large reasoning models on a single-turn paradigm with verifiable rewards. However, we observe that models trained with existing RL paradigms often lose their ability to solve problems across multiple turns and struggle to revise answers based on contextual feedback, leading to repetitive responses. We ask: can LRMs learn to reflect their answers in a multi-turn context? In this work, we find that trai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.14295","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.14295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.14295","created_at":"2026-07-05T11:57:38.047629+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.14295v2","created_at":"2026-07-05T11:57:38.047629+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.14295","created_at":"2026-07-05T11:57:38.047629+00:00"},{"alias_kind":"pith_short_12","alias_value":"QWUJIRD23WRV","created_at":"2026-07-05T11:57:38.047629+00:00"},{"alias_kind":"pith_short_16","alias_value":"QWUJIRD23WRVONMT","created_at":"2026-07-05T11:57:38.047629+00:00"},{"alias_kind":"pith_short_8","alias_value":"QWUJIRD2","created_at":"2026-07-05T11:57:38.047629+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31455","citing_title":"DRIFT: Decoupled Rollouts and Importance-Weighted Fine-Tuning for Efficient Multi-Turn Optimization","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O","json":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O.json","graph_json":"https://pith.science/api/pith-number/QWUJIRD23WRVONMTSGXMTICH2O/graph.json","events_json":"https://pith.science/api/pith-number/QWUJIRD23WRVONMTSGXMTICH2O/events.json","paper":"https://pith.science/paper/QWUJIRD2"},"agent_actions":{"view_html":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O","download_json":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O.json","view_paper":"https://pith.science/paper/QWUJIRD2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.14295&json=true","fetch_graph":"https://pith.science/api/pith-number/QWUJIRD23WRVONMTSGXMTICH2O/graph.json","fetch_events":"https://pith.science/api/pith-number/QWUJIRD23WRVONMTSGXMTICH2O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O/action/storage_attestation","attest_author":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O/action/author_attestation","sign_citation":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O/action/citation_signature","submit_replication":"https://pith.science/pith/QWUJIRD23WRVONMTSGXMTICH2O/action/replication_record"}},"created_at":"2026-07-05T11:57:38.047629+00:00","updated_at":"2026-07-05T11:57:38.047629+00:00"}