{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZHO3NIOGQRJ6Z56LLMSE3RTV2J","short_pith_number":"pith:ZHO3NIOG","schema_version":"1.0","canonical_sha256":"c9ddb6a1c68453ecf7cb5b244dc675d261bf552f1f2bc54ab7963bd1d48d61f5","source":{"kind":"arxiv","id":"2507.13158","version":1},"attestation_state":"computed","paper":{"title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hao Sun, Mihaela van der Schaar","submitted_at":"2025-07-17T14:22:24Z","abstract_excerpt":"In the era of Large Language Models (LLMs), alignment has emerged as a fundamental yet challenging problem in the pursuit of more reliable, controllable, and capable machine intelligence. The recent success of reasoning models and conversational AI systems has underscored the critical role of reinforcement learning (RL) in enhancing these systems, driving increased research interest at the intersection of RL and LLM alignment. This paper provides a comprehensive review of recent advances in LLM alignment through the lens of inverse reinforcement learning (IRL), emphasizing the distinctions bet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.13158","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-17T14:22:24Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7b97b8334bce5c8a711b314cccf165ead25dd33a8e5978f53fa4a3a7eb898fad","abstract_canon_sha256":"25e865df7b6da85a5b15e146627040dc050a38a250dd50a831fdf13b510971c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:55.668427Z","signature_b64":"IXbhciL1yfoz21mFLge9ir6K8fMl2Lk/PwoNLHUKfXY08csucpuOZRstKW5HbaZDW493tTkVOQ9dTodv4nLyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9ddb6a1c68453ecf7cb5b244dc675d261bf552f1f2bc54ab7963bd1d48d61f5","last_reissued_at":"2026-07-05T11:38:55.667905Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:55.667905Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hao Sun, Mihaela van der Schaar","submitted_at":"2025-07-17T14:22:24Z","abstract_excerpt":"In the era of Large Language Models (LLMs), alignment has emerged as a fundamental yet challenging problem in the pursuit of more reliable, controllable, and capable machine intelligence. The recent success of reasoning models and conversational AI systems has underscored the critical role of reinforcement learning (RL) in enhancing these systems, driving increased research interest at the intersection of RL and LLM alignment. This paper provides a comprehensive review of recent advances in LLM alignment through the lens of inverse reinforcement learning (IRL), emphasizing the distinctions bet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.13158","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.13158/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.13158","created_at":"2026-07-05T11:38:55.667966+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.13158v1","created_at":"2026-07-05T11:38:55.667966+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.13158","created_at":"2026-07-05T11:38:55.667966+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHO3NIOGQRJ6","created_at":"2026-07-05T11:38:55.667966+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHO3NIOGQRJ6Z56L","created_at":"2026-07-05T11:38:55.667966+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHO3NIOG","created_at":"2026-07-05T11:38:55.667966+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02259","citing_title":"The Alignment Flywheel: A Governance-Centric Hybrid MAS for Architecture-Agnostic Safety","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J","json":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J.json","graph_json":"https://pith.science/api/pith-number/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/graph.json","events_json":"https://pith.science/api/pith-number/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/events.json","paper":"https://pith.science/paper/ZHO3NIOG"},"agent_actions":{"view_html":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J","download_json":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J.json","view_paper":"https://pith.science/paper/ZHO3NIOG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.13158&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/action/storage_attestation","attest_author":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/action/author_attestation","sign_citation":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/action/citation_signature","submit_replication":"https://pith.science/pith/ZHO3NIOGQRJ6Z56LLMSE3RTV2J/action/replication_record"}},"created_at":"2026-07-05T11:38:55.667966+00:00","updated_at":"2026-07-05T11:38:55.667966+00:00"}