{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D6ND3DMTXEAAITDONZMOY46VTD","short_pith_number":"pith:D6ND3DMT","schema_version":"1.0","canonical_sha256":"1f9a3d8d93b900044c6e6e58ec73d598f82a9048f4eacbc820ae0fc1a3ed1f7b","source":{"kind":"arxiv","id":"2403.12017","version":1},"attestation_state":"computed","paper":{"title":"Supervised Fine-Tuning as Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hao Sun","submitted_at":"2024-03-18T17:52:57Z","abstract_excerpt":"The prevailing approach to aligning Large Language Models (LLMs) typically relies on human or AI feedback and assumes access to specific types of preference datasets. In our work, we question the efficacy of such datasets and explore various scenarios where alignment with expert demonstrations proves more realistic. We build a sequential decision-making framework to formulate the problem of aligning LLMs using demonstration datasets. Drawing insights from inverse reinforcement learning and imitation learning, we introduce various approaches for divergence minimization in the LLM alignment task"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.12017","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-18T17:52:57Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"9d42715de4c1f50eca414aa0a43f6bb218e1c1e8f276be7a754349ca0cf9b61c","abstract_canon_sha256":"792b9f23718a7254e38d0d4014e69f8a2f1a903e1e5e94e310ce18c173c11602"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:57:35.216591Z","signature_b64":"cu3FVFxATH6s3qnnIExBSf8G0AoK0KzaZdvGv01mnf/X9QCdpxcbW4645v62N/XrByvl2MI2EpWqqkhUJE0aDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f9a3d8d93b900044c6e6e58ec73d598f82a9048f4eacbc820ae0fc1a3ed1f7b","last_reissued_at":"2026-07-05T07:57:35.216081Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:57:35.216081Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Supervised Fine-Tuning as Inverse Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hao Sun","submitted_at":"2024-03-18T17:52:57Z","abstract_excerpt":"The prevailing approach to aligning Large Language Models (LLMs) typically relies on human or AI feedback and assumes access to specific types of preference datasets. In our work, we question the efficacy of such datasets and explore various scenarios where alignment with expert demonstrations proves more realistic. We build a sequential decision-making framework to formulate the problem of aligning LLMs using demonstration datasets. Drawing insights from inverse reinforcement learning and imitation learning, we introduce various approaches for divergence minimization in the LLM alignment task"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.12017","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.12017/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.12017","created_at":"2026-07-05T07:57:35.216143+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.12017v1","created_at":"2026-07-05T07:57:35.216143+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.12017","created_at":"2026-07-05T07:57:35.216143+00:00"},{"alias_kind":"pith_short_12","alias_value":"D6ND3DMTXEAA","created_at":"2026-07-05T07:57:35.216143+00:00"},{"alias_kind":"pith_short_16","alias_value":"D6ND3DMTXEAAITDO","created_at":"2026-07-05T07:57:35.216143+00:00"},{"alias_kind":"pith_short_8","alias_value":"D6ND3DMT","created_at":"2026-07-05T07:57:35.216143+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16995","citing_title":"SPS: Steering Probability Squeezing for Better Exploration in Reinforcement Learning for Large Language Models","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD","json":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD.json","graph_json":"https://pith.science/api/pith-number/D6ND3DMTXEAAITDONZMOY46VTD/graph.json","events_json":"https://pith.science/api/pith-number/D6ND3DMTXEAAITDONZMOY46VTD/events.json","paper":"https://pith.science/paper/D6ND3DMT"},"agent_actions":{"view_html":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD","download_json":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD.json","view_paper":"https://pith.science/paper/D6ND3DMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.12017&json=true","fetch_graph":"https://pith.science/api/pith-number/D6ND3DMTXEAAITDONZMOY46VTD/graph.json","fetch_events":"https://pith.science/api/pith-number/D6ND3DMTXEAAITDONZMOY46VTD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD/action/storage_attestation","attest_author":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD/action/author_attestation","sign_citation":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD/action/citation_signature","submit_replication":"https://pith.science/pith/D6ND3DMTXEAAITDONZMOY46VTD/action/replication_record"}},"created_at":"2026-07-05T07:57:35.216143+00:00","updated_at":"2026-07-05T07:57:35.216143+00:00"}