{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:MYCZXL5MNUDGV2DZZWUDIWSJL4","short_pith_number":"pith:MYCZXL5M","schema_version":"1.0","canonical_sha256":"66059bafac6d066ae879cda8345a495f28f9ece09695f85c8888b28817c33761","source":{"kind":"arxiv","id":"2608.03573","version":1},"attestation_state":"computed","paper":{"title":"SFT Conflicts, RL Coexists: A Theoretical and Empirical Analysis of Multi-Task Learning for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hongbang Yuan, Juanzi Li, Jun Zhao, Kang Liu, Kejian Zhu, Shangqing Tu, Yushi Bai, Zhuoran Jin","submitted_at":"2026-08-04T12:32:26Z","abstract_excerpt":"Supervised Fine-Tuning (SFT) and Reinforcement Learning (RL) exhibit fundamentally different behaviors in enhancing multi-task reasoning for large language models (LLMs). Our preliminary experiments revealed a phenomenon: SFT suffers from severe task conflicts under multi-stage training, whereas RL enables stable coexistence across diverse tasks. Empirically, we trace this to the parameter level, observing that RL induces sparse and approximately orthogonal updates across tasks. We provide a theoretical explanation for this mechanism by analyzing multi-task gradient interference. Our results r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.03573","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-04T12:32:26Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"077f03e6e38c0561258cdd6dd6bcc040184357f62302f95e397ceb018ab5842b","abstract_canon_sha256":"e7d4c4a10f7a676c47e2d6d7fd70b8dc30b1f5a6cd1d0ce4edf86c0d80bee797"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-05T01:36:40.403129Z","signature_b64":"Y8GBSxAu1F7HeZBKK8oV8B2oiKRrYS5Pxz7Is8fl1+PD2IfjkbaRwwmYwqWbkE3n5chbxyPz6b3JdaETfBQTDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66059bafac6d066ae879cda8345a495f28f9ece09695f85c8888b28817c33761","last_reissued_at":"2026-08-05T01:36:40.401556Z","signature_status":"signed_v1","first_computed_at":"2026-08-05T01:36:40.401556Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SFT Conflicts, RL Coexists: A Theoretical and Empirical Analysis of Multi-Task Learning for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Hongbang Yuan, Juanzi Li, Jun Zhao, Kang Liu, Kejian Zhu, Shangqing Tu, Yushi Bai, Zhuoran Jin","submitted_at":"2026-08-04T12:32:26Z","abstract_excerpt":"Supervised Fine-Tuning (SFT) and Reinforcement Learning (RL) exhibit fundamentally different behaviors in enhancing multi-task reasoning for large language models (LLMs). Our preliminary experiments revealed a phenomenon: SFT suffers from severe task conflicts under multi-stage training, whereas RL enables stable coexistence across diverse tasks. Empirically, we trace this to the parameter level, observing that RL induces sparse and approximately orthogonal updates across tasks. We provide a theoretical explanation for this mechanism by analyzing multi-task gradient interference. Our results r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.03573","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.03573/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.03573","created_at":"2026-08-05T01:36:40.402025+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.03573v1","created_at":"2026-08-05T01:36:40.402025+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.03573","created_at":"2026-08-05T01:36:40.402025+00:00"},{"alias_kind":"pith_short_12","alias_value":"MYCZXL5MNUDG","created_at":"2026-08-05T01:36:40.402025+00:00"},{"alias_kind":"pith_short_16","alias_value":"MYCZXL5MNUDGV2DZ","created_at":"2026-08-05T01:36:40.402025+00:00"},{"alias_kind":"pith_short_8","alias_value":"MYCZXL5M","created_at":"2026-08-05T01:36:40.402025+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4","json":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4.json","graph_json":"https://pith.science/api/pith-number/MYCZXL5MNUDGV2DZZWUDIWSJL4/graph.json","events_json":"https://pith.science/api/pith-number/MYCZXL5MNUDGV2DZZWUDIWSJL4/events.json","paper":"https://pith.science/paper/MYCZXL5M"},"agent_actions":{"view_html":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4","download_json":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4.json","view_paper":"https://pith.science/paper/MYCZXL5M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.03573&json=true","fetch_graph":"https://pith.science/api/pith-number/MYCZXL5MNUDGV2DZZWUDIWSJL4/graph.json","fetch_events":"https://pith.science/api/pith-number/MYCZXL5MNUDGV2DZZWUDIWSJL4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4/action/storage_attestation","attest_author":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4/action/author_attestation","sign_citation":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4/action/citation_signature","submit_replication":"https://pith.science/pith/MYCZXL5MNUDGV2DZZWUDIWSJL4/action/replication_record"}},"created_at":"2026-08-05T01:36:40.402025+00:00","updated_at":"2026-08-05T01:36:40.402025+00:00"}