{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:B3WCONJZYSYUYB3HHP4A2K2WBL","short_pith_number":"pith:B3WCONJZ","schema_version":"1.0","canonical_sha256":"0eec273539c4b14c07673bf80d2b560aeb311c84c43e556e66d8a95d54ce1a4d","source":{"kind":"arxiv","id":"2407.00626","version":2},"attestation_state":"computed","paper":{"title":"Maximum Entropy Inverse Reinforcement Learning of Diffusion Models with Energy-Based Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dohyun Kwon, Frank C. Park, Himchan Hwang, Sangwoong Yoon, Yung-Kyun Noh","submitted_at":"2024-06-30T08:52:17Z","abstract_excerpt":"We present a maximum entropy inverse reinforcement learning (IRL) approach for improving the sample quality of diffusion generative models, especially when the number of generation time steps is small. Similar to how IRL trains a policy based on the reward function learned from expert demonstrations, we train (or fine-tune) a diffusion model using the log probability density estimated from training data. Since we employ an energy-based model (EBM) to represent the log density, our approach boils down to the joint training of a diffusion model and an EBM. Our IRL formulation, named Diffusion by"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00626","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-30T08:52:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8dedc8b767b4c2b853f95889864411466dcc2d1383249371c73a401f21e1ad4c","abstract_canon_sha256":"5e90e88846082cb2d2ee16d5a91857ed1cf39e067730c2ad49f8018cba24e39c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:52.607912Z","signature_b64":"kaY+iBARUc4WVixD35qWR16IdhTJdvI1zOlAS0ZfTmiMFdiW9BN2OORyyM1ysBdHZpuaQrMSdBN6ldXKgpu9DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0eec273539c4b14c07673bf80d2b560aeb311c84c43e556e66d8a95d54ce1a4d","last_reissued_at":"2026-07-05T09:28:52.607396Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:52.607396Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Maximum Entropy Inverse Reinforcement Learning of Diffusion Models with Energy-Based Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dohyun Kwon, Frank C. Park, Himchan Hwang, Sangwoong Yoon, Yung-Kyun Noh","submitted_at":"2024-06-30T08:52:17Z","abstract_excerpt":"We present a maximum entropy inverse reinforcement learning (IRL) approach for improving the sample quality of diffusion generative models, especially when the number of generation time steps is small. Similar to how IRL trains a policy based on the reward function learned from expert demonstrations, we train (or fine-tune) a diffusion model using the log probability density estimated from training data. Since we employ an energy-based model (EBM) to represent the log density, our approach boils down to the joint training of a diffusion model and an EBM. Our IRL formulation, named Diffusion by"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00626","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00626","created_at":"2026-07-05T09:28:52.607457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00626v2","created_at":"2026-07-05T09:28:52.607457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00626","created_at":"2026-07-05T09:28:52.607457+00:00"},{"alias_kind":"pith_short_12","alias_value":"B3WCONJZYSYU","created_at":"2026-07-05T09:28:52.607457+00:00"},{"alias_kind":"pith_short_16","alias_value":"B3WCONJZYSYUYB3H","created_at":"2026-07-05T09:28:52.607457+00:00"},{"alias_kind":"pith_short_8","alias_value":"B3WCONJZ","created_at":"2026-07-05T09:28:52.607457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01819","citing_title":"Score as Action: Fine-Tuning Diffusion Generative Models by Continuous-time Reinforcement Learning","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL","json":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL.json","graph_json":"https://pith.science/api/pith-number/B3WCONJZYSYUYB3HHP4A2K2WBL/graph.json","events_json":"https://pith.science/api/pith-number/B3WCONJZYSYUYB3HHP4A2K2WBL/events.json","paper":"https://pith.science/paper/B3WCONJZ"},"agent_actions":{"view_html":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL","download_json":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL.json","view_paper":"https://pith.science/paper/B3WCONJZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00626&json=true","fetch_graph":"https://pith.science/api/pith-number/B3WCONJZYSYUYB3HHP4A2K2WBL/graph.json","fetch_events":"https://pith.science/api/pith-number/B3WCONJZYSYUYB3HHP4A2K2WBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL/action/storage_attestation","attest_author":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL/action/author_attestation","sign_citation":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL/action/citation_signature","submit_replication":"https://pith.science/pith/B3WCONJZYSYUYB3HHP4A2K2WBL/action/replication_record"}},"created_at":"2026-07-05T09:28:52.607457+00:00","updated_at":"2026-07-05T09:28:52.607457+00:00"}