{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U4J7KEVGBAV66GNA2UOKP2UFEU","short_pith_number":"pith:U4J7KEVG","schema_version":"1.0","canonical_sha256":"a713f512a6082bef19a0d51ca7ea8525028b66a9f3abb85600eb455843677676","source":{"kind":"arxiv","id":"2501.14400","version":2},"attestation_state":"computed","paper":{"title":"SKIL: Semantic Keypoint Imitation Learning for Generalizable Data-efficient Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Jiacheng You, Jiongye Li, Shengjie Wang, Yang Gao, Yihang Hu","submitted_at":"2025-01-24T11:11:53Z","abstract_excerpt":"Real-world tasks such as garment manipulation and table rearrangement demand robots to perform generalizable, highly precise, and long-horizon actions. Although imitation learning has proven to be an effective approach for teaching robots new skills, large amounts of expert demonstration data are still indispensible for these complex tasks, resulting in high sample complexity and costly data collection. To address this, we propose Semantic Keypoint Imitation Learning (SKIL), a framework which automatically obtains semantic keypoints with the help of vision foundation models, and forms the desc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14400","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-01-24T11:11:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3daf3bf9271e99cf907c095a18ce39bd6011d5fa992eb0e2d058f793779f2961","abstract_canon_sha256":"21b7479288274b0c202eed250b953aa6a8d54cfaf20019ae90aa2469897c9cec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:40.952675Z","signature_b64":"N6telbOC527LfZDBz4WknfIzGs0ECGXp25QTDnSn8tBR23wN/iChNJlIAwnvmu81Xzv6tHz5Lou0dCpugLC0Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a713f512a6082bef19a0d51ca7ea8525028b66a9f3abb85600eb455843677676","last_reissued_at":"2026-07-05T11:30:40.952149Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:40.952149Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SKIL: Semantic Keypoint Imitation Learning for Generalizable Data-efficient Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Jiacheng You, Jiongye Li, Shengjie Wang, Yang Gao, Yihang Hu","submitted_at":"2025-01-24T11:11:53Z","abstract_excerpt":"Real-world tasks such as garment manipulation and table rearrangement demand robots to perform generalizable, highly precise, and long-horizon actions. Although imitation learning has proven to be an effective approach for teaching robots new skills, large amounts of expert demonstration data are still indispensible for these complex tasks, resulting in high sample complexity and costly data collection. To address this, we propose Semantic Keypoint Imitation Learning (SKIL), a framework which automatically obtains semantic keypoints with the help of vision foundation models, and forms the desc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14400","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14400/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14400","created_at":"2026-07-05T11:30:40.952210+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14400v2","created_at":"2026-07-05T11:30:40.952210+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14400","created_at":"2026-07-05T11:30:40.952210+00:00"},{"alias_kind":"pith_short_12","alias_value":"U4J7KEVGBAV6","created_at":"2026-07-05T11:30:40.952210+00:00"},{"alias_kind":"pith_short_16","alias_value":"U4J7KEVGBAV66GNA","created_at":"2026-07-05T11:30:40.952210+00:00"},{"alias_kind":"pith_short_8","alias_value":"U4J7KEVG","created_at":"2026-07-05T11:30:40.952210+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24338","citing_title":"RoBoSR: Structured Scene Representations for Embodied Robotic Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00836","citing_title":"From World Models to World Action Models: A Concise Tutorial for Robotics","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13515","citing_title":"MaskWAM: Unifying Mask Prompting and Prediction for World-Action Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00836","citing_title":"From World Models to World Action Models: A Concise Tutorial for Robotics","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26649","citing_title":"On the Generalization Capabilities, Design Choices and Limitations of Keypoint Imitation Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15944","citing_title":"FocalPolicy: Frequency-Optimized Chunking and Locally Anchored Flow Matching for Coherent Visuomotor Policy","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15836","citing_title":"GAP: Geometric Anchor Pre-training for Data-Efficient Visuomotor Learning of Manipulation Tasks","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15944","citing_title":"FocalPolicy: Frequency-Optimized Chunking and Locally Anchored Flow Matching for Coherent Visuomotor Policy","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17601","citing_title":"From a Single Demonstration to a General Policy for Contact-Rich Manipulation","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15215","citing_title":"A Hierarchical Spatiotemporal Action Tokenizer for In-Context Imitation Learning in Robotics","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13428","citing_title":"SID: Sliding into Distribution for Robust Few-Demonstration Manipulation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10201","citing_title":"HeteroGenManip: Generalizable Manipulation For Heterogeneous Object Interactions","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10201","citing_title":"HeteroGenManip: Generalizable Manipulation For Heterogeneous Object Interactions","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15215","citing_title":"A Hierarchical Spatiotemporal Action Tokenizer for In-Context Imitation Learning in Robotics","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU","json":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU.json","graph_json":"https://pith.science/api/pith-number/U4J7KEVGBAV66GNA2UOKP2UFEU/graph.json","events_json":"https://pith.science/api/pith-number/U4J7KEVGBAV66GNA2UOKP2UFEU/events.json","paper":"https://pith.science/paper/U4J7KEVG"},"agent_actions":{"view_html":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU","download_json":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU.json","view_paper":"https://pith.science/paper/U4J7KEVG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14400&json=true","fetch_graph":"https://pith.science/api/pith-number/U4J7KEVGBAV66GNA2UOKP2UFEU/graph.json","fetch_events":"https://pith.science/api/pith-number/U4J7KEVGBAV66GNA2UOKP2UFEU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU/action/storage_attestation","attest_author":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU/action/author_attestation","sign_citation":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU/action/citation_signature","submit_replication":"https://pith.science/pith/U4J7KEVGBAV66GNA2UOKP2UFEU/action/replication_record"}},"created_at":"2026-07-05T11:30:40.952210+00:00","updated_at":"2026-07-05T11:30:40.952210+00:00"}