{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZCEOSKKE3WINS2HB73TPNOFF5H","short_pith_number":"pith:ZCEOSKKE","schema_version":"1.0","canonical_sha256":"c888e92944dd90d968e1fee6f6b8a5e9ec613f632b5995bd1381c8fbaf9c25c8","source":{"kind":"arxiv","id":"2312.08533","version":4},"attestation_state":"computed","paper":{"title":"World Models via Policy-Guided Trajectory Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ingmar Posner, Jun Yamada, Marc Rigter","submitted_at":"2023-12-13T21:46:09Z","abstract_excerpt":"World models are a powerful tool for developing intelligent agents. By predicting the outcome of a sequence of actions, world models enable policies to be optimised via on-policy reinforcement learning (RL) using synthetic data, i.e. in \"in imagination\". Existing world models are autoregressive in that they interleave predicting the next state with sampling the next action from the policy. Prediction error inevitably compounds as the trajectory length grows. In this work, we propose a novel world modelling approach that is not autoregressive and generates entire on-policy trajectories in a sin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.08533","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-13T21:46:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5aa8faddab70d654189009dbb3476ca40e6a480ddb194930ec6d62ee17cd662f","abstract_canon_sha256":"1ad3e4999258cacb69fad55cc195c7edaad8d01763eea5cde387e88013784131"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:01:17.670991Z","signature_b64":"8uo0mvnNtQaN0xk6y/B5GANCj8UZXb2/aB9xeYTx3yTJbvWPAcoE8vdw3x3d/FAjMk8qQWWqVyNdcBZrC4omCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c888e92944dd90d968e1fee6f6b8a5e9ec613f632b5995bd1381c8fbaf9c25c8","last_reissued_at":"2026-07-05T08:01:17.670438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:01:17.670438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"World Models via Policy-Guided Trajectory Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ingmar Posner, Jun Yamada, Marc Rigter","submitted_at":"2023-12-13T21:46:09Z","abstract_excerpt":"World models are a powerful tool for developing intelligent agents. By predicting the outcome of a sequence of actions, world models enable policies to be optimised via on-policy reinforcement learning (RL) using synthetic data, i.e. in \"in imagination\". Existing world models are autoregressive in that they interleave predicting the next state with sampling the next action from the policy. Prediction error inevitably compounds as the trajectory length grows. In this work, we propose a novel world modelling approach that is not autoregressive and generates entire on-policy trajectories in a sin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.08533","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.08533/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.08533","created_at":"2026-07-05T08:01:17.670516+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.08533v4","created_at":"2026-07-05T08:01:17.670516+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.08533","created_at":"2026-07-05T08:01:17.670516+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZCEOSKKE3WIN","created_at":"2026-07-05T08:01:17.670516+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZCEOSKKE3WINS2HB","created_at":"2026-07-05T08:01:17.670516+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZCEOSKKE","created_at":"2026-07-05T08:01:17.670516+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.03828","citing_title":"From Static Constraints to Dynamic Adaptation: Sample-Level Constraint Relaxation for Offline-to-Online Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":182,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19538","citing_title":"DAWM: Diffusion Action World Models for Offline Reinforcement Learning via Action-Inferred Transitions","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03828","citing_title":"From Static Constraints to Dynamic Adaptation: Sample-Level Constraint Relaxation for Offline-to-Online Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04812","citing_title":"Multimodal Diffusion Forcing for Forceful Manipulation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09035","citing_title":"Advantage-Guided Diffusion for Model-Based Reinforcement Learning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H","json":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H.json","graph_json":"https://pith.science/api/pith-number/ZCEOSKKE3WINS2HB73TPNOFF5H/graph.json","events_json":"https://pith.science/api/pith-number/ZCEOSKKE3WINS2HB73TPNOFF5H/events.json","paper":"https://pith.science/paper/ZCEOSKKE"},"agent_actions":{"view_html":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H","download_json":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H.json","view_paper":"https://pith.science/paper/ZCEOSKKE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.08533&json=true","fetch_graph":"https://pith.science/api/pith-number/ZCEOSKKE3WINS2HB73TPNOFF5H/graph.json","fetch_events":"https://pith.science/api/pith-number/ZCEOSKKE3WINS2HB73TPNOFF5H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H/action/storage_attestation","attest_author":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H/action/author_attestation","sign_citation":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H/action/citation_signature","submit_replication":"https://pith.science/pith/ZCEOSKKE3WINS2HB73TPNOFF5H/action/replication_record"}},"created_at":"2026-07-05T08:01:17.670516+00:00","updated_at":"2026-07-05T08:01:17.670516+00:00"}