{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BRBRMJRLJZXG2SP5CRWPOTWTOO","short_pith_number":"pith:BRBRMJRL","schema_version":"1.0","canonical_sha256":"0c4316262b4e6e6d49fd146cf74ed373902037618d2532c1ee7188098ecd75e3","source":{"kind":"arxiv","id":"2301.10677","version":2},"attestation_state":"computed","paper":{"title":"Imitating Human Behaviour with Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Anssi Kanervisto, Dave Bignell, Ida Momennejad, Katja Hofmann, Mingfei Sun, Raluca Georgescu, Sam Devlin, Sergio Valcarcel Macua, Shan Zheng Tan, Tabish Rashid, Tim Pearce","submitted_at":"2023-01-25T16:31:05Z","abstract_excerpt":"Diffusion models have emerged as powerful generative models in the text-to-image domain. This paper studies their application as observation-to-action models for imitating human behaviour in sequential environments. Human behaviour is stochastic and multimodal, with structured correlations between action dimensions. Meanwhile, standard modelling choices in behaviour cloning are limited in their expressiveness and may introduce bias into the cloned policy. We begin by pointing out the limitations of these choices. We then propose that diffusion models are an excellent fit for imitating human be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.10677","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-01-25T16:31:05Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"d84dbb6ba4a58ec127ee377c2ad14bc11043cc9de46492d890963ff46900a61d","abstract_canon_sha256":"d9be015c2c4b7d9b7f1c8359ae91ecceddde28967d35e4ecbc0bda68eb213b5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:47:47.394324Z","signature_b64":"llpz5Gtv5CbyYA6dnGuFEr8PhG1AstqIVlG6pn332fwHyMdfQE0o76kpKdmApj4uFpI+pOJXjHOTfj9WFgdZAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c4316262b4e6e6d49fd146cf74ed373902037618d2532c1ee7188098ecd75e3","last_reissued_at":"2026-07-05T05:47:47.393803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:47:47.393803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Imitating Human Behaviour with Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Anssi Kanervisto, Dave Bignell, Ida Momennejad, Katja Hofmann, Mingfei Sun, Raluca Georgescu, Sam Devlin, Sergio Valcarcel Macua, Shan Zheng Tan, Tabish Rashid, Tim Pearce","submitted_at":"2023-01-25T16:31:05Z","abstract_excerpt":"Diffusion models have emerged as powerful generative models in the text-to-image domain. This paper studies their application as observation-to-action models for imitating human behaviour in sequential environments. Human behaviour is stochastic and multimodal, with structured correlations between action dimensions. Meanwhile, standard modelling choices in behaviour cloning are limited in their expressiveness and may introduce bias into the cloned policy. We begin by pointing out the limitations of these choices. We then propose that diffusion models are an excellent fit for imitating human be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.10677","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.10677/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.10677","created_at":"2026-07-05T05:47:47.393864+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.10677v2","created_at":"2026-07-05T05:47:47.393864+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.10677","created_at":"2026-07-05T05:47:47.393864+00:00"},{"alias_kind":"pith_short_12","alias_value":"BRBRMJRLJZXG","created_at":"2026-07-05T05:47:47.393864+00:00"},{"alias_kind":"pith_short_16","alias_value":"BRBRMJRLJZXG2SP5","created_at":"2026-07-05T05:47:47.393864+00:00"},{"alias_kind":"pith_short_8","alias_value":"BRBRMJRL","created_at":"2026-07-05T05:47:47.393864+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21935","citing_title":"CoRDE: Concept-Prior Routed Diffusion Experts for Structural Generalization in Robot Manipulation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15148","citing_title":"MimicIK: Real-Time Generative Inverse Kinematics from Teleoperation with FK Consistency","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12109","citing_title":"InDex: Empowering VLA Models with Intent-Conditioned Arm-Hand Coordination for Dexterous Manipulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06712","citing_title":"Data-Efficient Autoregressive-to-Diffusion Language Models via On-Policy Distillation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06418","citing_title":"Double Preconditioning (DoPr): Optimization for Test-Time Performance, not Validation Loss","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27095","citing_title":"Adversarial Dual On-Policy Distillation from Expressive Teacher","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27919","citing_title":"Frequency-Guided Action Diffusion via Sub-Frequency Manifold Traversal","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01151","citing_title":"Lagrangian Perturbation Diffusion Steering: Latent Reinforcement Learning for Generative Policies","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12978","citing_title":"Learning Native Continuation for Action Chunking Flow Policies","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20299","citing_title":"Mechanisms of Misgeneralization in Physical Sequence Modeling","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10631","citing_title":"HybridVLA: Collaborative Diffusion and Autoregression in a Unified Vision-Language-Action Model","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2404.12377","citing_title":"RoboDreamer: Learning Compositional World Models for Robot Imagination","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07339","citing_title":"Real-Time Execution of Action Chunking Flow Policies","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2208.06193","citing_title":"Diffusion Policies as an Expressive Policy Class for Offline Reinforcement Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13316","citing_title":"Test-time Sparsity for Extreme Fast Action Diffusion","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03404","citing_title":"Diffusion Policy with Bayesian Expert Selection for Active Multi-Target Tracking","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12416","citing_title":"Aligning Flow Map Policies with Optimal Q-Guidance","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2411.19650","citing_title":"CogACT: A Foundational Vision-Language-Action Model for Synergizing Cognition and Action in Robotic Manipulation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21331","citing_title":"FingerViP: Learning Real-World Dexterous Manipulation with Fingertip Visual Perception","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07319","citing_title":"Generative Modeling with Flux Matching","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06168","citing_title":"Action Images: End-to-End Policy Learning via Multiview Video Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03065","citing_title":"OGPO: Sample Efficient Full-Finetuning of Generative Control Policies","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO","json":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO.json","graph_json":"https://pith.science/api/pith-number/BRBRMJRLJZXG2SP5CRWPOTWTOO/graph.json","events_json":"https://pith.science/api/pith-number/BRBRMJRLJZXG2SP5CRWPOTWTOO/events.json","paper":"https://pith.science/paper/BRBRMJRL"},"agent_actions":{"view_html":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO","download_json":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO.json","view_paper":"https://pith.science/paper/BRBRMJRL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.10677&json=true","fetch_graph":"https://pith.science/api/pith-number/BRBRMJRLJZXG2SP5CRWPOTWTOO/graph.json","fetch_events":"https://pith.science/api/pith-number/BRBRMJRLJZXG2SP5CRWPOTWTOO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO/action/storage_attestation","attest_author":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO/action/author_attestation","sign_citation":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO/action/citation_signature","submit_replication":"https://pith.science/pith/BRBRMJRLJZXG2SP5CRWPOTWTOO/action/replication_record"}},"created_at":"2026-07-05T05:47:47.393864+00:00","updated_at":"2026-07-05T05:47:47.393864+00:00"}