{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GADUKDWMPMTNI72BW4Y7YSKMC6","short_pith_number":"pith:GADUKDWM","schema_version":"1.0","canonical_sha256":"3007450ecc7b26d47f41b731fc494c17a743d5581f6a9883d9511db71bac14a5","source":{"kind":"arxiv","id":"2403.19578","version":3},"attestation_state":"computed","paper":{"title":"Keypoint Action Tokens Enable In-Context Imitation Learning in Robotics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.NE"],"primary_cat":"cs.RO","authors_text":"Edward Johns, Norman Di Palo","submitted_at":"2024-03-28T17:04:00Z","abstract_excerpt":"We show that off-the-shelf text-based Transformers, with no additional training, can perform few-shot in-context visual imitation learning, mapping visual observations to action sequences that emulate the demonstrator's behaviour. We achieve this by transforming visual observations (inputs) and trajectories of actions (outputs) into sequences of tokens that a text-pretrained Transformer (GPT-4 Turbo) can ingest and generate, via a framework we call Keypoint Action Tokens (KAT). Despite being trained only on language, we show that these Transformers excel at translating tokenised visual keypoin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.19578","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-03-28T17:04:00Z","cross_cats_sorted":["cs.LG","cs.NE"],"title_canon_sha256":"b6d801b782626f5559130fd017c58fa743caf317a5e0cf0c3a1f73164b1fc590","abstract_canon_sha256":"64f54971b6489b4c55fef0a51c7cb5350eba01fad1a34a703a505f2ae3f22037"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:22:17.083630Z","signature_b64":"Hcxjz1Dcg3U+PqcQWim2EjGqeEEVFEGIKamgv1wNNuUg66fJHCfcQ1OIQBzwxKMcbLrruLLil+hLdCEoRPmzCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3007450ecc7b26d47f41b731fc494c17a743d5581f6a9883d9511db71bac14a5","last_reissued_at":"2026-07-05T09:22:17.083147Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:22:17.083147Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Keypoint Action Tokens Enable In-Context Imitation Learning in Robotics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.NE"],"primary_cat":"cs.RO","authors_text":"Edward Johns, Norman Di Palo","submitted_at":"2024-03-28T17:04:00Z","abstract_excerpt":"We show that off-the-shelf text-based Transformers, with no additional training, can perform few-shot in-context visual imitation learning, mapping visual observations to action sequences that emulate the demonstrator's behaviour. We achieve this by transforming visual observations (inputs) and trajectories of actions (outputs) into sequences of tokens that a text-pretrained Transformer (GPT-4 Turbo) can ingest and generate, via a framework we call Keypoint Action Tokens (KAT). Despite being trained only on language, we show that these Transformers excel at translating tokenised visual keypoin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19578","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19578/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.19578","created_at":"2026-07-05T09:22:17.083209+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.19578v3","created_at":"2026-07-05T09:22:17.083209+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19578","created_at":"2026-07-05T09:22:17.083209+00:00"},{"alias_kind":"pith_short_12","alias_value":"GADUKDWMPMTN","created_at":"2026-07-05T09:22:17.083209+00:00"},{"alias_kind":"pith_short_16","alias_value":"GADUKDWMPMTNI72B","created_at":"2026-07-05T09:22:17.083209+00:00"},{"alias_kind":"pith_short_8","alias_value":"GADUKDWM","created_at":"2026-07-05T09:22:17.083209+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26801","citing_title":"Improving Vision-Language-Action Model Fine-Tuning with Structured Stage and Keyframe Supervision","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08154","citing_title":"SynthICL: Scalable In-context Imitation Learning with Synthetic Data","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05952","citing_title":"Learning of Robot Safety Policies via Adversarial Synthetic Scenarios","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2504.04991","citing_title":"Wavelet Policy: Imitation Learning in the Scale Domain with World Prior Memory","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.00306","citing_title":"J-PARSE: Jacobian-based Projection Algorithm for Resolving Singularities Effectively in Inverse Kinematic Control of Serial Manipulators","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15215","citing_title":"A Hierarchical Spatiotemporal Action Tokenizer for In-Context Imitation Learning in Robotics","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2409.01652","citing_title":"ReKep: Spatio-Temporal Reasoning of Relational Keypoint Constraints for Robotic Manipulation","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01448","citing_title":"Decompose and Recompose: Reasoning New Skills from Existing Abilities for Cross-Task Robotic Manipulation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15215","citing_title":"A Hierarchical Spatiotemporal Action Tokenizer for In-Context Imitation Learning in Robotics","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6","json":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6.json","graph_json":"https://pith.science/api/pith-number/GADUKDWMPMTNI72BW4Y7YSKMC6/graph.json","events_json":"https://pith.science/api/pith-number/GADUKDWMPMTNI72BW4Y7YSKMC6/events.json","paper":"https://pith.science/paper/GADUKDWM"},"agent_actions":{"view_html":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6","download_json":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6.json","view_paper":"https://pith.science/paper/GADUKDWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.19578&json=true","fetch_graph":"https://pith.science/api/pith-number/GADUKDWMPMTNI72BW4Y7YSKMC6/graph.json","fetch_events":"https://pith.science/api/pith-number/GADUKDWMPMTNI72BW4Y7YSKMC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6/action/storage_attestation","attest_author":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6/action/author_attestation","sign_citation":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6/action/citation_signature","submit_replication":"https://pith.science/pith/GADUKDWMPMTNI72BW4Y7YSKMC6/action/replication_record"}},"created_at":"2026-07-05T09:22:17.083209+00:00","updated_at":"2026-07-05T09:22:17.083209+00:00"}