{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5RG5WHVY23DZYQSCZPY2UVEZRT","short_pith_number":"pith:5RG5WHVY","schema_version":"1.0","canonical_sha256":"ec4ddb1eb8d6c79c4242cbf1aa54998cefa1e198f3831aced670c0531c8725bf","source":{"kind":"arxiv","id":"2410.12456","version":2},"attestation_state":"computed","paper":{"title":"Training Neural Samplers with Reverse Diffusive KL Divergence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"David Barber, Jiajun He, Jos\\'e Miguel Hern\\'andez-Lobato, Mingtian Zhang, Wenlin Chen","submitted_at":"2024-10-16T11:08:02Z","abstract_excerpt":"Training generative models to sample from unnormalized density functions is an important and challenging task in machine learning. Traditional training methods often rely on the reverse Kullback-Leibler (KL) divergence due to its tractability. However, the mode-seeking behavior of reverse KL hinders effective approximation of multi-modal target distributions. To address this, we propose to minimize the reverse KL along diffusion trajectories of both model and target densities. We refer to this objective as the reverse diffusive KL divergence, which allows the model to capture multiple modes. L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12456","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-16T11:08:02Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"a7a7c6f8305d64623dfabe2ca4a136e3db02bc2de73a183d6e184b083ebe07c0","abstract_canon_sha256":"29dad6087d05bea3cc71878268b57723d75c28a590a580d69c64469dcc946577"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:24.301068Z","signature_b64":"S4aqNaH62UsKNpO83tnOAthHiA+4pUMOUAUyhUpQaFLFkD4GFAPQboHBlDSsnkaj+ER+bQBnULELqtm85qcqCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec4ddb1eb8d6c79c4242cbf1aa54998cefa1e198f3831aced670c0531c8725bf","last_reissued_at":"2026-07-05T10:23:24.300212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:24.300212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Neural Samplers with Reverse Diffusive KL Divergence","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"David Barber, Jiajun He, Jos\\'e Miguel Hern\\'andez-Lobato, Mingtian Zhang, Wenlin Chen","submitted_at":"2024-10-16T11:08:02Z","abstract_excerpt":"Training generative models to sample from unnormalized density functions is an important and challenging task in machine learning. Traditional training methods often rely on the reverse Kullback-Leibler (KL) divergence due to its tractability. However, the mode-seeking behavior of reverse KL hinders effective approximation of multi-modal target distributions. To address this, we propose to minimize the reverse KL along diffusion trajectories of both model and target densities. We refer to this objective as the reverse diffusive KL divergence, which allows the model to capture multiple modes. L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12456","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12456/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12456","created_at":"2026-07-05T10:23:24.300324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12456v2","created_at":"2026-07-05T10:23:24.300324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12456","created_at":"2026-07-05T10:23:24.300324+00:00"},{"alias_kind":"pith_short_12","alias_value":"5RG5WHVY23DZ","created_at":"2026-07-05T10:23:24.300324+00:00"},{"alias_kind":"pith_short_16","alias_value":"5RG5WHVY23DZYQSC","created_at":"2026-07-05T10:23:24.300324+00:00"},{"alias_kind":"pith_short_8","alias_value":"5RG5WHVY","created_at":"2026-07-05T10:23:24.300324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18478","citing_title":"Data-Forcing Distillation: Restoring Diversity and Fidelity in Few-Step Video Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03726","citing_title":"Energy-Weighted Flow Matching: Unlocking Continuous Normalizing Flows for Efficient and Scalable Boltzmann Sampling","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05710","citing_title":"On the Blessing of Pre-training in Weak-to-Strong Generalization","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT","json":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT.json","graph_json":"https://pith.science/api/pith-number/5RG5WHVY23DZYQSCZPY2UVEZRT/graph.json","events_json":"https://pith.science/api/pith-number/5RG5WHVY23DZYQSCZPY2UVEZRT/events.json","paper":"https://pith.science/paper/5RG5WHVY"},"agent_actions":{"view_html":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT","download_json":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT.json","view_paper":"https://pith.science/paper/5RG5WHVY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12456&json=true","fetch_graph":"https://pith.science/api/pith-number/5RG5WHVY23DZYQSCZPY2UVEZRT/graph.json","fetch_events":"https://pith.science/api/pith-number/5RG5WHVY23DZYQSCZPY2UVEZRT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT/action/storage_attestation","attest_author":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT/action/author_attestation","sign_citation":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT/action/citation_signature","submit_replication":"https://pith.science/pith/5RG5WHVY23DZYQSCZPY2UVEZRT/action/replication_record"}},"created_at":"2026-07-05T10:23:24.300324+00:00","updated_at":"2026-07-05T10:23:24.300324+00:00"}