{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7P3DAJLSFD4LYSZTIBTYLAIWCC","short_pith_number":"pith:7P3DAJLS","schema_version":"1.0","canonical_sha256":"fbf630257228f8bc4b33406785811610836d259cd46cc925acfdb33b4eb7c0fd","source":{"kind":"arxiv","id":"2209.05451","version":2},"attestation_state":"computed","paper":{"title":"Perceiver-Actor: A Multi-Task Transformer for Robotic Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Dieter Fox, Lucas Manuelli, Mohit Shridhar","submitted_at":"2022-09-12T17:51:05Z","abstract_excerpt":"Transformers have revolutionized vision and natural language processing with their ability to scale with large datasets. But in robotic manipulation, data is both limited and expensive. Can manipulation still benefit from Transformers with the right problem formulation? We investigate this question with PerAct, a language-conditioned behavior-cloning agent for multi-task 6-DoF manipulation. PerAct encodes language goals and RGB-D voxel observations with a Perceiver Transformer, and outputs discretized actions by ``detecting the next best voxel action''. Unlike frameworks that operate on 2D ima"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.05451","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-12T17:51:05Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV","cs.LG"],"title_canon_sha256":"f5f44842f7e6ca0d04bdf3b6a60e0a6f3250bbfad2ce71eea6a3679306b315e7","abstract_canon_sha256":"0250cd4d3f453cd889b8bbca8c85b73480597f58be0d9ea4b11950acf46dee38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:15:12.450837Z","signature_b64":"Q3Hf3PRrLd7W03sZJSY8zNK6xwsU0d71MFK5Wm4aeYTml42SGmIMwDB7FLkMbEFDbdZO1ZAubhvQ1oAageSVCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fbf630257228f8bc4b33406785811610836d259cd46cc925acfdb33b4eb7c0fd","last_reissued_at":"2026-07-05T05:15:12.450302Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:15:12.450302Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Perceiver-Actor: A Multi-Task Transformer for Robotic Manipulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Dieter Fox, Lucas Manuelli, Mohit Shridhar","submitted_at":"2022-09-12T17:51:05Z","abstract_excerpt":"Transformers have revolutionized vision and natural language processing with their ability to scale with large datasets. But in robotic manipulation, data is both limited and expensive. Can manipulation still benefit from Transformers with the right problem formulation? We investigate this question with PerAct, a language-conditioned behavior-cloning agent for multi-task 6-DoF manipulation. PerAct encodes language goals and RGB-D voxel observations with a Perceiver Transformer, and outputs discretized actions by ``detecting the next best voxel action''. Unlike frameworks that operate on 2D ima"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.05451","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.05451/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.05451","created_at":"2026-07-05T05:15:12.450399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.05451v2","created_at":"2026-07-05T05:15:12.450399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.05451","created_at":"2026-07-05T05:15:12.450399+00:00"},{"alias_kind":"pith_short_12","alias_value":"7P3DAJLSFD4L","created_at":"2026-07-05T05:15:12.450399+00:00"},{"alias_kind":"pith_short_16","alias_value":"7P3DAJLSFD4LYSZT","created_at":"2026-07-05T05:15:12.450399+00:00"},{"alias_kind":"pith_short_8","alias_value":"7P3DAJLS","created_at":"2026-07-05T05:15:12.450399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23625","citing_title":"Learning to See While Learning to Act: Diffusion Models for Active Perception in Robot Imitation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02092","citing_title":"Guided Action Flow: Q-Guided Inference for Flow-Matching Vision-Language-Action Policies","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02735","citing_title":"See Less, Specify More: Visual Evidence Budgets for Generalizable VLAs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29864","citing_title":"LLM-Guided Future Hypotheses for Horizon-Aware Exploration in Multi-Step Robot Manipulation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14135","citing_title":"GAF: Gaussian Action Field as a 4D Representation for Dynamic World Modeling in Robotic Manipulation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2302.11550","citing_title":"Scaling Robot Learning with Semantically Imagined Experience","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05117","citing_title":"SeedPolicy: Horizon Scaling via Self-Evolving Diffusion Policy for Robot Manipulation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02117","citing_title":"Mobile ALOHA: Learning Bimanual Mobile Manipulation with Low-Cost Whole-Body Teleoperation","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06481","citing_title":"OA-WAM: Object-Addressable World Action Model for Robust Robot Manipulation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2304.13705","citing_title":"Learning Fine-Grained Bimanual Manipulation with Low-Cost Hardware","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2303.03378","citing_title":"PaLM-E: An Embodied Multimodal Language Model","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC","json":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC.json","graph_json":"https://pith.science/api/pith-number/7P3DAJLSFD4LYSZTIBTYLAIWCC/graph.json","events_json":"https://pith.science/api/pith-number/7P3DAJLSFD4LYSZTIBTYLAIWCC/events.json","paper":"https://pith.science/paper/7P3DAJLS"},"agent_actions":{"view_html":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC","download_json":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC.json","view_paper":"https://pith.science/paper/7P3DAJLS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.05451&json=true","fetch_graph":"https://pith.science/api/pith-number/7P3DAJLSFD4LYSZTIBTYLAIWCC/graph.json","fetch_events":"https://pith.science/api/pith-number/7P3DAJLSFD4LYSZTIBTYLAIWCC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC/action/storage_attestation","attest_author":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC/action/author_attestation","sign_citation":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC/action/citation_signature","submit_replication":"https://pith.science/pith/7P3DAJLSFD4LYSZTIBTYLAIWCC/action/replication_record"}},"created_at":"2026-07-05T05:15:12.450399+00:00","updated_at":"2026-07-05T05:15:12.450399+00:00"}