{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WTCXU5CRVTKT7W5BYOBPZ7S6CI","short_pith_number":"pith:WTCXU5CR","schema_version":"1.0","canonical_sha256":"b4c57a7451acd53fdba1c382fcfe5e1237a0ffbe160203e9c3c6177b730027f0","source":{"kind":"arxiv","id":"2505.17988","version":3},"attestation_state":"computed","paper":{"title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jiandong Gao, Ji Wu, Yutong Chen","submitted_at":"2025-05-23T14:55:22Z","abstract_excerpt":"R1-style Reinforcement Learning (RL) significantly enhances Large Language Models' reasoning capabilities, yet the mechanism behind rule-based RL remains unclear. We found that small-scale SFT has substantial influence on RL but shows poor efficiency. To explain our observations, we propose an analytical framework and compare the efficiency of SFT and RL by measuring \\textbf{sample effect}. Our hypothetical analysis shows the potential to improve SFT efficiency. Guided by our analysis, we propose \\textbf{Re-distillation}, a technique that aims to boost the effectiveness of small-scale distilla"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17988","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-23T14:55:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"858c22a5087b6a0d69bba40ba79f90130883359f59b4a0c39c249d6e6e8885a9","abstract_canon_sha256":"d91a0b7d24722712901f4630944614e0f8e9b5e73385dec2c26067281c7618bb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:49.190497Z","signature_b64":"C078/kguGuFE5BgsOLq48ck4B0DXYfe1objJzfvvGC0DNbbNV4JIx4XMWQC8T2nTrlR7oecXCkYSWu9kpGMDDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b4c57a7451acd53fdba1c382fcfe5e1237a0ffbe160203e9c3c6177b730027f0","last_reissued_at":"2026-07-05T11:48:49.189880Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:49.189880Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Revealing the Effectiveness of Small-Scale Fine-tuning in R1-style Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jiandong Gao, Ji Wu, Yutong Chen","submitted_at":"2025-05-23T14:55:22Z","abstract_excerpt":"R1-style Reinforcement Learning (RL) significantly enhances Large Language Models' reasoning capabilities, yet the mechanism behind rule-based RL remains unclear. We found that small-scale SFT has substantial influence on RL but shows poor efficiency. To explain our observations, we propose an analytical framework and compare the efficiency of SFT and RL by measuring \\textbf{sample effect}. Our hypothetical analysis shows the potential to improve SFT efficiency. Guided by our analysis, we propose \\textbf{Re-distillation}, a technique that aims to boost the effectiveness of small-scale distilla"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17988","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17988/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17988","created_at":"2026-07-05T11:48:49.189931+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17988v3","created_at":"2026-07-05T11:48:49.189931+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17988","created_at":"2026-07-05T11:48:49.189931+00:00"},{"alias_kind":"pith_short_12","alias_value":"WTCXU5CRVTKT","created_at":"2026-07-05T11:48:49.189931+00:00"},{"alias_kind":"pith_short_16","alias_value":"WTCXU5CRVTKT7W5B","created_at":"2026-07-05T11:48:49.189931+00:00"},{"alias_kind":"pith_short_8","alias_value":"WTCXU5CR","created_at":"2026-07-05T11:48:49.189931+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.13579","citing_title":"Toward Better EHR Reasoning in LLMs: Reinforcement Learning with Expert Attention Guidance","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI","json":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI.json","graph_json":"https://pith.science/api/pith-number/WTCXU5CRVTKT7W5BYOBPZ7S6CI/graph.json","events_json":"https://pith.science/api/pith-number/WTCXU5CRVTKT7W5BYOBPZ7S6CI/events.json","paper":"https://pith.science/paper/WTCXU5CR"},"agent_actions":{"view_html":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI","download_json":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI.json","view_paper":"https://pith.science/paper/WTCXU5CR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17988&json=true","fetch_graph":"https://pith.science/api/pith-number/WTCXU5CRVTKT7W5BYOBPZ7S6CI/graph.json","fetch_events":"https://pith.science/api/pith-number/WTCXU5CRVTKT7W5BYOBPZ7S6CI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI/action/storage_attestation","attest_author":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI/action/author_attestation","sign_citation":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI/action/citation_signature","submit_replication":"https://pith.science/pith/WTCXU5CRVTKT7W5BYOBPZ7S6CI/action/replication_record"}},"created_at":"2026-07-05T11:48:49.189931+00:00","updated_at":"2026-07-05T11:48:49.189931+00:00"}