{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CHP6E2IALEUORKBOXZDQEHQK7Z","short_pith_number":"pith:CHP6E2IA","schema_version":"1.0","canonical_sha256":"11dfe269005928e8a82ebe47021e0afe774d408cbec846292a82dea6349fbe4f","source":{"kind":"arxiv","id":"2411.15382","version":2},"attestation_state":"computed","paper":{"title":"On the Impact of Fine-Tuning on Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chirag Agarwal, Elita Lobo, Himabindu Lakkaraju","submitted_at":"2024-11-22T23:54:37Z","abstract_excerpt":"Large language models have emerged as powerful tools for general intelligence, showcasing advanced natural language processing capabilities that find applications across diverse domains. Despite their impressive performance, recent studies have highlighted the potential for significant enhancements in LLMs' task-specific performance through fine-tuning strategies like Reinforcement Learning with Human Feedback (RLHF), supervised fine-tuning (SFT), and Quantized Low-Rank Adapters (Q-LoRA) method. However, previous works have shown that while fine-tuning offers significant performance gains, it "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15382","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-22T23:54:37Z","cross_cats_sorted":[],"title_canon_sha256":"665500fef286d5a9d97a51ea012f876b4d10a0590af907f5d08b00243fa662fa","abstract_canon_sha256":"a6240ee763a59060e9fa2baf25c8171e0ae902872d09bb9b2263239e3b430f5c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:50.559328Z","signature_b64":"UsW22LI8NXP43JMzT8AnUv+POXDJ7KNjlc/kOzAE8wCLvADwAdx/AvmPJ16X4Kf26hdVvNPLkmnoM62ihxqTDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11dfe269005928e8a82ebe47021e0afe774d408cbec846292a82dea6349fbe4f","last_reissued_at":"2026-07-05T10:41:50.558810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:50.558810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Impact of Fine-Tuning on Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chirag Agarwal, Elita Lobo, Himabindu Lakkaraju","submitted_at":"2024-11-22T23:54:37Z","abstract_excerpt":"Large language models have emerged as powerful tools for general intelligence, showcasing advanced natural language processing capabilities that find applications across diverse domains. Despite their impressive performance, recent studies have highlighted the potential for significant enhancements in LLMs' task-specific performance through fine-tuning strategies like Reinforcement Learning with Human Feedback (RLHF), supervised fine-tuning (SFT), and Quantized Low-Rank Adapters (Q-LoRA) method. However, previous works have shown that while fine-tuning offers significant performance gains, it "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15382","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15382/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15382","created_at":"2026-07-05T10:41:50.558876+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15382v2","created_at":"2026-07-05T10:41:50.558876+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15382","created_at":"2026-07-05T10:41:50.558876+00:00"},{"alias_kind":"pith_short_12","alias_value":"CHP6E2IALEUO","created_at":"2026-07-05T10:41:50.558876+00:00"},{"alias_kind":"pith_short_16","alias_value":"CHP6E2IALEUORKBO","created_at":"2026-07-05T10:41:50.558876+00:00"},{"alias_kind":"pith_short_8","alias_value":"CHP6E2IA","created_at":"2026-07-05T10:41:50.558876+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05199","citing_title":"Reason, Reward, Refine: Step-Level Errors Corrections with Structured Feedback for Physics Reasoning in Small Language Models","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2501.09686","citing_title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11549","citing_title":"UNIPO: Unified Interactive Visual Explanation for RL Fine-Tuning Policy Optimization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07138","citing_title":"Can You Break RLVER? Probing Adversarial Robustness of RL-Trained Empathetic Agents","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08299","citing_title":"SeLaR: Selective Latent Reasoning in Large Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05504","citing_title":"Semantic Communication with an LLM-enabled Knowledge Base","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z","json":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z.json","graph_json":"https://pith.science/api/pith-number/CHP6E2IALEUORKBOXZDQEHQK7Z/graph.json","events_json":"https://pith.science/api/pith-number/CHP6E2IALEUORKBOXZDQEHQK7Z/events.json","paper":"https://pith.science/paper/CHP6E2IA"},"agent_actions":{"view_html":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z","download_json":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z.json","view_paper":"https://pith.science/paper/CHP6E2IA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15382&json=true","fetch_graph":"https://pith.science/api/pith-number/CHP6E2IALEUORKBOXZDQEHQK7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/CHP6E2IALEUORKBOXZDQEHQK7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z/action/storage_attestation","attest_author":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z/action/author_attestation","sign_citation":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z/action/citation_signature","submit_replication":"https://pith.science/pith/CHP6E2IALEUORKBOXZDQEHQK7Z/action/replication_record"}},"created_at":"2026-07-05T10:41:50.558876+00:00","updated_at":"2026-07-05T10:41:50.558876+00:00"}