{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FNIAIRF2G7Y6CRWWOKVGLAY6OF","short_pith_number":"pith:FNIAIRF2","schema_version":"1.0","canonical_sha256":"2b500444ba37f1e146d672aa65831e7154d09bd51ba0bb684d39606a0a815665","source":{"kind":"arxiv","id":"2501.10105","version":2},"attestation_state":"computed","paper":{"title":"Universal Actions for Enhanced Embodied Foundation Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Dongxiu Liu, Jianxiong Li, Jingjing Liu, Jinliang Zheng, Xianyuan Zhan, Ya-Qin Zhang, Yinan Zheng, Yu Liu, Zhihao Wang, Zhonghong Ou","submitted_at":"2025-01-17T10:45:22Z","abstract_excerpt":"Training on diverse, internet-scale data is a key factor in the success of recent large foundation models. Yet, using the same recipe for building embodied agents has faced noticeable difficulties. Despite the availability of many crowd-sourced embodied datasets, their action spaces often exhibit significant heterogeneity due to distinct physical embodiment and control interfaces for different robots, causing substantial challenges in developing embodied foundation models using cross-domain data. In this paper, we introduce UniAct, a new embodied foundation modeling framework operating in a Un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.10105","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.RO","submitted_at":"2025-01-17T10:45:22Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"f03ffe72be616853f68e3fe47efe579c522990e65ab854ecc0d4cab1cf91b911","abstract_canon_sha256":"ae73c943c76a9b3cf234e13453e7265ddcbac61fb5b6fd95dfe69874305e5624"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:43.215148Z","signature_b64":"8ufZTJbN5HDi0EZ40JKn3yanz8s8rxFjS8khLZh+XnBpZoJY4C9Y/HBwgOzMeWS/qC2127UbfP6lOGu31yeJAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b500444ba37f1e146d672aa65831e7154d09bd51ba0bb684d39606a0a815665","last_reissued_at":"2026-07-05T10:26:43.214406Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:43.214406Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Universal Actions for Enhanced Embodied Foundation Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Dongxiu Liu, Jianxiong Li, Jingjing Liu, Jinliang Zheng, Xianyuan Zhan, Ya-Qin Zhang, Yinan Zheng, Yu Liu, Zhihao Wang, Zhonghong Ou","submitted_at":"2025-01-17T10:45:22Z","abstract_excerpt":"Training on diverse, internet-scale data is a key factor in the success of recent large foundation models. Yet, using the same recipe for building embodied agents has faced noticeable difficulties. Despite the availability of many crowd-sourced embodied datasets, their action spaces often exhibit significant heterogeneity due to distinct physical embodiment and control interfaces for different robots, causing substantial challenges in developing embodied foundation models using cross-domain data. In this paper, we introduce UniAct, a new embodied foundation modeling framework operating in a Un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.10105","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.10105/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.10105","created_at":"2026-07-05T10:26:43.214489+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.10105v2","created_at":"2026-07-05T10:26:43.214489+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.10105","created_at":"2026-07-05T10:26:43.214489+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNIAIRF2G7Y6","created_at":"2026-07-05T10:26:43.214489+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNIAIRF2G7Y6CRWW","created_at":"2026-07-05T10:26:43.214489+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNIAIRF2","created_at":"2026-07-05T10:26:43.214489+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.19417","citing_title":"Hi Robot: Open-Ended Instruction Following with Hierarchical Vision-Language-Action Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2501.15830","citing_title":"SpatialVLA: Exploring Spatial Representations for Visual-Language-Action Model","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24622","citing_title":"CF-VLA: Efficient Coarse-to-Fine Action Generation for Vision-Language-Action Policies","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06192","citing_title":"EA-WM: Event-Aware Generative World Model with Structured Kinematic-to-Visual Action Fields","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20100","citing_title":"JoyAI-RA 0.1: A Foundation Model for Robotic Autonomy","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF","json":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF.json","graph_json":"https://pith.science/api/pith-number/FNIAIRF2G7Y6CRWWOKVGLAY6OF/graph.json","events_json":"https://pith.science/api/pith-number/FNIAIRF2G7Y6CRWWOKVGLAY6OF/events.json","paper":"https://pith.science/paper/FNIAIRF2"},"agent_actions":{"view_html":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF","download_json":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF.json","view_paper":"https://pith.science/paper/FNIAIRF2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.10105&json=true","fetch_graph":"https://pith.science/api/pith-number/FNIAIRF2G7Y6CRWWOKVGLAY6OF/graph.json","fetch_events":"https://pith.science/api/pith-number/FNIAIRF2G7Y6CRWWOKVGLAY6OF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF/action/storage_attestation","attest_author":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF/action/author_attestation","sign_citation":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF/action/citation_signature","submit_replication":"https://pith.science/pith/FNIAIRF2G7Y6CRWWOKVGLAY6OF/action/replication_record"}},"created_at":"2026-07-05T10:26:43.214489+00:00","updated_at":"2026-07-05T10:26:43.214489+00:00"}