{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QYACGAA4K3UKYCVQSZYNFZXKBU","short_pith_number":"pith:QYACGAA4","schema_version":"1.0","canonical_sha256":"860023001c56e8ac0ab09670d2e6ea0d27c19076e62e000c5c578fad556c24bb","source":{"kind":"arxiv","id":"2503.02505","version":2},"attestation_state":"computed","paper":{"title":"ROCKET-2: Steering Visuomotor Policy via Cross-View Goal Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.RO"],"primary_cat":"cs.AI","authors_text":"Anji Liu, Shaofei Cai, Yitao Liang, Zhancun Mu","submitted_at":"2025-03-04T11:16:46Z","abstract_excerpt":"We aim to develop a goal specification method that is semantically clear, spatially sensitive, domain-agnostic, and intuitive for human users to guide agent interactions in 3D environments. Specifically, we propose a novel cross-view goal alignment framework that allows users to specify target objects using segmentation masks from their camera views rather than the agent's observations. We highlight that behavior cloning alone fails to align the agent's behavior with human intent when the human and agent camera views differ significantly. To address this, we introduce two auxiliary objectives:"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.02505","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-03-04T11:16:46Z","cross_cats_sorted":["cs.CV","cs.LG","cs.RO"],"title_canon_sha256":"2e3f42b5dae0fc625956c63561cde08b569373781d9616e812e60083940aba02","abstract_canon_sha256":"b94ad37419be0aaf0bc412f6dcd2fa305479762e52fbb727e410009ef8cf8314"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:05.196557Z","signature_b64":"RcGnI55lCjibU9SdWVLAX4Hi6MLIE4d28saryL3ZdsiYmO6uTusXa0yIqI/7frDGisbcbRWfxUfgjRD81aRlAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"860023001c56e8ac0ab09670d2e6ea0d27c19076e62e000c5c578fad556c24bb","last_reissued_at":"2026-07-05T11:34:05.196110Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:05.196110Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ROCKET-2: Steering Visuomotor Policy via Cross-View Goal Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.RO"],"primary_cat":"cs.AI","authors_text":"Anji Liu, Shaofei Cai, Yitao Liang, Zhancun Mu","submitted_at":"2025-03-04T11:16:46Z","abstract_excerpt":"We aim to develop a goal specification method that is semantically clear, spatially sensitive, domain-agnostic, and intuitive for human users to guide agent interactions in 3D environments. Specifically, we propose a novel cross-view goal alignment framework that allows users to specify target objects using segmentation masks from their camera views rather than the agent's observations. We highlight that behavior cloning alone fails to align the agent's behavior with human intent when the human and agent camera views differ significantly. To address this, we introduce two auxiliary objectives:"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.02505","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.02505/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.02505","created_at":"2026-07-05T11:34:05.196166+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.02505v2","created_at":"2026-07-05T11:34:05.196166+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.02505","created_at":"2026-07-05T11:34:05.196166+00:00"},{"alias_kind":"pith_short_12","alias_value":"QYACGAA4K3UK","created_at":"2026-07-05T11:34:05.196166+00:00"},{"alias_kind":"pith_short_16","alias_value":"QYACGAA4K3UKYCVQ","created_at":"2026-07-05T11:34:05.196166+00:00"},{"alias_kind":"pith_short_8","alias_value":"QYACGAA4","created_at":"2026-07-05T11:34:05.196166+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01848","citing_title":"RescueBench: Can Embodied Agents Save Lives in the Wild ?","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17537","citing_title":"Self-supervised Hierarchical Visual Reasoning with World Model","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17537","citing_title":"Self-supervised Hierarchical Visual Reasoning with World Model","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU","json":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU.json","graph_json":"https://pith.science/api/pith-number/QYACGAA4K3UKYCVQSZYNFZXKBU/graph.json","events_json":"https://pith.science/api/pith-number/QYACGAA4K3UKYCVQSZYNFZXKBU/events.json","paper":"https://pith.science/paper/QYACGAA4"},"agent_actions":{"view_html":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU","download_json":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU.json","view_paper":"https://pith.science/paper/QYACGAA4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.02505&json=true","fetch_graph":"https://pith.science/api/pith-number/QYACGAA4K3UKYCVQSZYNFZXKBU/graph.json","fetch_events":"https://pith.science/api/pith-number/QYACGAA4K3UKYCVQSZYNFZXKBU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU/action/storage_attestation","attest_author":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU/action/author_attestation","sign_citation":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU/action/citation_signature","submit_replication":"https://pith.science/pith/QYACGAA4K3UKYCVQSZYNFZXKBU/action/replication_record"}},"created_at":"2026-07-05T11:34:05.196166+00:00","updated_at":"2026-07-05T11:34:05.196166+00:00"}