{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QKI7HZJXMQW5TLX2I75M2SHK2C","short_pith_number":"pith:QKI7HZJX","schema_version":"1.0","canonical_sha256":"8291f3e537642dd9aefa47facd48ead0b6ffab54d784158b84fdee9101696443","source":{"kind":"arxiv","id":"2504.21530","version":1},"attestation_state":"computed","paper":{"title":"RoboGround: Robotic Manipulation with Grounded Vision-Language Priors","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Haifeng Huang, Hao Li, Jiangmiao Pang, Tai Wang, Xiaoshen Han, Xinyi Chen, Yilun Chen, Zehan Wang, Zhou Zhao","submitted_at":"2025-04-30T11:26:40Z","abstract_excerpt":"Recent advancements in robotic manipulation have highlighted the potential of intermediate representations for improving policy generalization. In this work, we explore grounding masks as an effective intermediate representation, balancing two key advantages: (1) effective spatial guidance that specifies target objects and placement areas while also conveying information about object shape and size, and (2) broad generalization potential driven by large-scale vision-language models pretrained on diverse grounding datasets. We introduce RoboGround, a grounding-aware robotic manipulation system "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.21530","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-04-30T11:26:40Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"742def69ae3976a6ffd0b939ea3fdf613fed1f8c582f6222211484d245538bbe","abstract_canon_sha256":"42f9a4fd3dcf94645a57cd4d0e534c32ed97b620fb4c142c0dfde93d87fc8dcb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:25.302872Z","signature_b64":"ArNNCJuYVG8k0FiBLyPUz6+v+eFpQfIdJg/A7gzYdplK6cmIBKiB/7aPzGgTsbEn60/ERYiNQ23fkzQC0ESmCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8291f3e537642dd9aefa47facd48ead0b6ffab54d784158b84fdee9101696443","last_reissued_at":"2026-07-05T10:56:25.302372Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:25.302372Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RoboGround: Robotic Manipulation with Grounded Vision-Language Priors","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Haifeng Huang, Hao Li, Jiangmiao Pang, Tai Wang, Xiaoshen Han, Xinyi Chen, Yilun Chen, Zehan Wang, Zhou Zhao","submitted_at":"2025-04-30T11:26:40Z","abstract_excerpt":"Recent advancements in robotic manipulation have highlighted the potential of intermediate representations for improving policy generalization. In this work, we explore grounding masks as an effective intermediate representation, balancing two key advantages: (1) effective spatial guidance that specifies target objects and placement areas while also conveying information about object shape and size, and (2) broad generalization potential driven by large-scale vision-language models pretrained on diverse grounding datasets. We introduce RoboGround, a grounding-aware robotic manipulation system "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.21530","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.21530/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.21530","created_at":"2026-07-05T10:56:25.302435+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.21530v1","created_at":"2026-07-05T10:56:25.302435+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.21530","created_at":"2026-07-05T10:56:25.302435+00:00"},{"alias_kind":"pith_short_12","alias_value":"QKI7HZJXMQW5","created_at":"2026-07-05T10:56:25.302435+00:00"},{"alias_kind":"pith_short_16","alias_value":"QKI7HZJXMQW5TLX2","created_at":"2026-07-05T10:56:25.302435+00:00"},{"alias_kind":"pith_short_8","alias_value":"QKI7HZJX","created_at":"2026-07-05T10:56:25.302435+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.12374","citing_title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C","json":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C.json","graph_json":"https://pith.science/api/pith-number/QKI7HZJXMQW5TLX2I75M2SHK2C/graph.json","events_json":"https://pith.science/api/pith-number/QKI7HZJXMQW5TLX2I75M2SHK2C/events.json","paper":"https://pith.science/paper/QKI7HZJX"},"agent_actions":{"view_html":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C","download_json":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C.json","view_paper":"https://pith.science/paper/QKI7HZJX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.21530&json=true","fetch_graph":"https://pith.science/api/pith-number/QKI7HZJXMQW5TLX2I75M2SHK2C/graph.json","fetch_events":"https://pith.science/api/pith-number/QKI7HZJXMQW5TLX2I75M2SHK2C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C/action/storage_attestation","attest_author":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C/action/author_attestation","sign_citation":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C/action/citation_signature","submit_replication":"https://pith.science/pith/QKI7HZJXMQW5TLX2I75M2SHK2C/action/replication_record"}},"created_at":"2026-07-05T10:56:25.302435+00:00","updated_at":"2026-07-05T10:56:25.302435+00:00"}