{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VEMHIRYYLQ4C55ROOSTSZ676QQ","short_pith_number":"pith:VEMHIRYY","schema_version":"1.0","canonical_sha256":"a9187447185c382ef62e74a72cfbfe84164ace8710ec9078ac11e9420a80eb52","source":{"kind":"arxiv","id":"2409.14411","version":2},"attestation_state":"computed","paper":{"title":"Scaling Diffusion Policy in Transformer to 1 Billion Parameters for Robotic Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chaomin Shen, Feifei Feng, Jian Tang, Jinming Li, Junjie Wen, Minjie Zhu, Ning Liu, Ran Cheng, Yaxin Peng, Yichen Zhu, Zhiyuan Xu","submitted_at":"2024-09-22T12:14:16Z","abstract_excerpt":"Diffusion Policy is a powerful technique tool for learning end-to-end visuomotor robot control. It is expected that Diffusion Policy possesses scalability, a key attribute for deep neural networks, typically suggesting that increasing model size would lead to enhanced performance. However, our observations indicate that Diffusion Policy in transformer architecture (\\DP) struggles to scale effectively; even minor additions of layers can deteriorate training outcomes. To address this issue, we introduce Scalable Diffusion Transformer Policy for visuomotor learning. Our proposed method, namely \\t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.14411","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-09-22T12:14:16Z","cross_cats_sorted":[],"title_canon_sha256":"ac2caf676f291ada32ed10b10e00f3d0b793a73d971f9d38c131cce4258a4f68","abstract_canon_sha256":"85269bd4341f1641ce59fdcd81994cd4449b3fac5e943e9e590c33ed750a0009"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:12.908088Z","signature_b64":"PPrF1DM59zTY8eziajjiS7sNeJLOPZllrauLZ9apl9R1FTkrylfJRiwWG6olwXgEYZ5wqv7BotE94b+9ai8nBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a9187447185c382ef62e74a72cfbfe84164ace8710ec9078ac11e9420a80eb52","last_reissued_at":"2026-07-05T09:35:12.907591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:12.907591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Diffusion Policy in Transformer to 1 Billion Parameters for Robotic Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chaomin Shen, Feifei Feng, Jian Tang, Jinming Li, Junjie Wen, Minjie Zhu, Ning Liu, Ran Cheng, Yaxin Peng, Yichen Zhu, Zhiyuan Xu","submitted_at":"2024-09-22T12:14:16Z","abstract_excerpt":"Diffusion Policy is a powerful technique tool for learning end-to-end visuomotor robot control. It is expected that Diffusion Policy possesses scalability, a key attribute for deep neural networks, typically suggesting that increasing model size would lead to enhanced performance. However, our observations indicate that Diffusion Policy in transformer architecture (\\DP) struggles to scale effectively; even minor additions of layers can deteriorate training outcomes. To address this issue, we introduce Scalable Diffusion Transformer Policy for visuomotor learning. Our proposed method, namely \\t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14411","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14411/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.14411","created_at":"2026-07-05T09:35:12.907654+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.14411v2","created_at":"2026-07-05T09:35:12.907654+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14411","created_at":"2026-07-05T09:35:12.907654+00:00"},{"alias_kind":"pith_short_12","alias_value":"VEMHIRYYLQ4C","created_at":"2026-07-05T09:35:12.907654+00:00"},{"alias_kind":"pith_short_16","alias_value":"VEMHIRYYLQ4C55RO","created_at":"2026-07-05T09:35:12.907654+00:00"},{"alias_kind":"pith_short_8","alias_value":"VEMHIRYY","created_at":"2026-07-05T09:35:12.907654+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06018","citing_title":"RoboTALES: Learning Reasoning-Guided Robot Policies via Task-Aligned Simulated Futures","ref_index":59,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10371","citing_title":"Test-time Adversarial Takeover: A Real-time Hijacking Interface against Robotic Diffusion Policies","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05855","citing_title":"DexVLA: Vision-Language Model with Plug-In Diffusion Expert for General Robot Control","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2410.24164","citing_title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ","json":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ.json","graph_json":"https://pith.science/api/pith-number/VEMHIRYYLQ4C55ROOSTSZ676QQ/graph.json","events_json":"https://pith.science/api/pith-number/VEMHIRYYLQ4C55ROOSTSZ676QQ/events.json","paper":"https://pith.science/paper/VEMHIRYY"},"agent_actions":{"view_html":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ","download_json":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ.json","view_paper":"https://pith.science/paper/VEMHIRYY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.14411&json=true","fetch_graph":"https://pith.science/api/pith-number/VEMHIRYYLQ4C55ROOSTSZ676QQ/graph.json","fetch_events":"https://pith.science/api/pith-number/VEMHIRYYLQ4C55ROOSTSZ676QQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ/action/storage_attestation","attest_author":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ/action/author_attestation","sign_citation":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ/action/citation_signature","submit_replication":"https://pith.science/pith/VEMHIRYYLQ4C55ROOSTSZ676QQ/action/replication_record"}},"created_at":"2026-07-05T09:35:12.907654+00:00","updated_at":"2026-07-05T09:35:12.907654+00:00"}