{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZSKGITFO4HYAU426J2YNRTICKE","short_pith_number":"pith:ZSKGITFO","schema_version":"1.0","canonical_sha256":"cc94644caee1f00a735e4eb0d8cd025137b75b9fd322441e6c8cd3d52a127c94","source":{"kind":"arxiv","id":"2501.18564","version":4},"attestation_state":"computed","paper":{"title":"SAM2Act: Integrating Visual Foundation Model with A Memory Architecture for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Dieter Fox, Haoquan Fang, Jiafei Duan, Markus Grotz, Ranjay Krishna, Wilbert Pumacay, Yi Ru Wang","submitted_at":"2025-01-30T18:37:16Z","abstract_excerpt":"Robotic manipulation systems operating in diverse, dynamic environments must exhibit three critical abilities: multitask interaction, generalization to unseen scenarios, and spatial memory. While significant progress has been made in robotic manipulation, existing approaches often fall short in generalization to complex environmental variations and addressing memory-dependent tasks. To bridge this gap, we introduce SAM2Act, a multi-view robotic transformer-based policy that leverages multi-resolution upsampling with visual representations from large-scale foundation model. SAM2Act achieves a s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.18564","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-01-30T18:37:16Z","cross_cats_sorted":[],"title_canon_sha256":"52ad23156f2f8f7f4925e6035b1e2c12b679abb6ee725f87cb284a8df90e6440","abstract_canon_sha256":"4d259bfec81f80c28c01f5ab64e4bca225e4af82fdf7d1330abbcd193487b7cf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:27.644781Z","signature_b64":"8AzvAM2JyqC2JR8dZQCWqRi+SRDUw4biGTsJvkG0lpgTYgNyIvutjjB/s2TPWpOhFpEARmoKoapRKvYzfJmRDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc94644caee1f00a735e4eb0d8cd025137b75b9fd322441e6c8cd3d52a127c94","last_reissued_at":"2026-07-05T11:36:27.644347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:27.644347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAM2Act: Integrating Visual Foundation Model with A Memory Architecture for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Dieter Fox, Haoquan Fang, Jiafei Duan, Markus Grotz, Ranjay Krishna, Wilbert Pumacay, Yi Ru Wang","submitted_at":"2025-01-30T18:37:16Z","abstract_excerpt":"Robotic manipulation systems operating in diverse, dynamic environments must exhibit three critical abilities: multitask interaction, generalization to unseen scenarios, and spatial memory. While significant progress has been made in robotic manipulation, existing approaches often fall short in generalization to complex environmental variations and addressing memory-dependent tasks. To bridge this gap, we introduce SAM2Act, a multi-view robotic transformer-based policy that leverages multi-resolution upsampling with visual representations from large-scale foundation model. SAM2Act achieves a s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.18564","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.18564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.18564","created_at":"2026-07-05T11:36:27.644407+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.18564v4","created_at":"2026-07-05T11:36:27.644407+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.18564","created_at":"2026-07-05T11:36:27.644407+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZSKGITFO4HYA","created_at":"2026-07-05T11:36:27.644407+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZSKGITFO4HYAU426","created_at":"2026-07-05T11:36:27.644407+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZSKGITFO","created_at":"2026-07-05T11:36:27.644407+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06678","citing_title":"NativeMEM: Native Memory Compression for Long-Horizon Robotic Manipulation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25136","citing_title":"Memory Retrieval in Visuomotor Policies for Long-Horizon Robot Control","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26443","citing_title":"WatchAct: A Benchmark for Behavior-Grounded Robot Manipulation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23589","citing_title":"KEMO: Event-Driven Keyframe Memory for Long-Horizon Robot Manipulation with VLA Policies","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21188","citing_title":"Remember what you did?: Learning Behavioral Memories for Partially Observable Object Manipulation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20092","citing_title":"EventVLA: Event-Driven Visual Evidence Memory for Long-Horizon Vision-Language-Action Policies","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12499","citing_title":"Action-Effect Memory Pretraining for Robot Manipulation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10363","citing_title":"HiMem-WAM: Hierarchical Memory-Gated World Action Models for Robotic Manipulation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04172","citing_title":"Affordance2Action: Task-Conditioned Scene-level Affordance Grounding for Real-Time Manipulation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30877","citing_title":"Wall-OSS-0.5 Technical Report","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30318","citing_title":"Chronos: A Physics-Informed Full-History Framework for Non-Markovian Long-Horizon Manipulation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20092","citing_title":"EventVLA: Event-Driven Visual Evidence Memory for Long-Horizon Vision-Language-Action Policies","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27677","citing_title":"DIM-WAM: World-Action Modeling with Diverse Historical Event Memory","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.02818","citing_title":"RoboMD: Uncovering Robot Vulnerabilities through Semantic Potential Fields","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23617","citing_title":"One Trajectory, One Token: Grounded Video Tokenization via Panoptic Sub-object Trajectory","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20323","citing_title":"PhysMem: Scaling Test-Time Memory for Embodied Physical Reasoning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03269","citing_title":"RLDX-1 Technical Report","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01448","citing_title":"Decompose and Recompose: Reasoning New Skills from Existing Abilities for Cross-Task Robotic Manipulation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18933","citing_title":"Gated Memory Policy: In-Context Memorization and Adaptation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03269","citing_title":"RLDX-1 Technical Report","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE","json":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE.json","graph_json":"https://pith.science/api/pith-number/ZSKGITFO4HYAU426J2YNRTICKE/graph.json","events_json":"https://pith.science/api/pith-number/ZSKGITFO4HYAU426J2YNRTICKE/events.json","paper":"https://pith.science/paper/ZSKGITFO"},"agent_actions":{"view_html":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE","download_json":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE.json","view_paper":"https://pith.science/paper/ZSKGITFO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.18564&json=true","fetch_graph":"https://pith.science/api/pith-number/ZSKGITFO4HYAU426J2YNRTICKE/graph.json","fetch_events":"https://pith.science/api/pith-number/ZSKGITFO4HYAU426J2YNRTICKE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE/action/storage_attestation","attest_author":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE/action/author_attestation","sign_citation":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE/action/citation_signature","submit_replication":"https://pith.science/pith/ZSKGITFO4HYAU426J2YNRTICKE/action/replication_record"}},"created_at":"2026-07-05T11:36:27.644407+00:00","updated_at":"2026-07-05T11:36:27.644407+00:00"}