{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7IN57SP6LBLQZVDKRNMLDBN7U3","short_pith_number":"pith:7IN57SP6","schema_version":"1.0","canonical_sha256":"fa1bdfc9fe58570cd46a8b58b185bfa6e236b6289f2951ea186b756c88fccc43","source":{"kind":"arxiv","id":"2412.20451","version":2},"attestation_state":"computed","paper":{"title":"CoA-VLA: Improving Vision-Language-Action Models via Visual-Textual Chain-of-Affordance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chengmeng Li, Feifei Feng, Jinming Li, Junjie Wen, Minjie Zhu, Ran Cheng, Xiaoyu Liu, Yan Peng, Yaxin Peng, Yichen Zhu, Zhibin Tang","submitted_at":"2024-12-29T12:24:31Z","abstract_excerpt":"Robot foundation models, particularly Vision-Language-Action (VLA) models, have garnered significant attention for their ability to enhance robot policy learning, greatly improving robot's generalization and robustness. OpenAI's recent model, O1, showcased impressive capabilities in solving complex problems by utilizing extensive reasoning chains. This prompts an important question: can robot models achieve better performance in multi-task , complex environments by reviewing prior observations and then providing task-specific reasoning to guide action prediction? In this paper, we introduce Ch"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.20451","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-12-29T12:24:31Z","cross_cats_sorted":[],"title_canon_sha256":"40ad2e06c7c9b5c2b3a5d767c623dee4a2309e06b55d49290b146d79825630db","abstract_canon_sha256":"30b880432a77c3d8ac74692abbbac52f36b66dc1f1442a40a3e1af84fd30d42a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:20.692743Z","signature_b64":"JAHBFvp6XyhBZooF344aYgKG/1wjrBL1kG/qMwSVA2f+sU46J8CME1hvVAkTpo7PslQL4WjHqjvVc1enLx6EDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa1bdfc9fe58570cd46a8b58b185bfa6e236b6289f2951ea186b756c88fccc43","last_reissued_at":"2026-07-05T11:46:20.692251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:20.692251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CoA-VLA: Improving Vision-Language-Action Models via Visual-Textual Chain-of-Affordance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chengmeng Li, Feifei Feng, Jinming Li, Junjie Wen, Minjie Zhu, Ran Cheng, Xiaoyu Liu, Yan Peng, Yaxin Peng, Yichen Zhu, Zhibin Tang","submitted_at":"2024-12-29T12:24:31Z","abstract_excerpt":"Robot foundation models, particularly Vision-Language-Action (VLA) models, have garnered significant attention for their ability to enhance robot policy learning, greatly improving robot's generalization and robustness. OpenAI's recent model, O1, showcased impressive capabilities in solving complex problems by utilizing extensive reasoning chains. This prompts an important question: can robot models achieve better performance in multi-task , complex environments by reviewing prior observations and then providing task-specific reasoning to guide action prediction? In this paper, we introduce Ch"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.20451","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.20451/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.20451","created_at":"2026-07-05T11:46:20.692305+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.20451v2","created_at":"2026-07-05T11:46:20.692305+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.20451","created_at":"2026-07-05T11:46:20.692305+00:00"},{"alias_kind":"pith_short_12","alias_value":"7IN57SP6LBLQ","created_at":"2026-07-05T11:46:20.692305+00:00"},{"alias_kind":"pith_short_16","alias_value":"7IN57SP6LBLQZVDK","created_at":"2026-07-05T11:46:20.692305+00:00"},{"alias_kind":"pith_short_8","alias_value":"7IN57SP6","created_at":"2026-07-05T11:46:20.692305+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06155","citing_title":"AffordanceVLA: A Vision-Language-Action Model Empowering Action Generation through Affordance-Aware Understanding","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00998","citing_title":"GraspGen-X: Cross-Embodiment 6-DOF Diffusion-based Grasping","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00110","citing_title":"General Covariant Action Modeling: Constructing Generalized Manifolds via Spatio-Temporal Decoupling","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22003","citing_title":"VP-VLA: Visual Prompting as an Interface for Vision-Language-Action Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05714","citing_title":"TriRelVLA: Triadic Relational Structure for Generalizable Embodied Manipulation","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3","json":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3.json","graph_json":"https://pith.science/api/pith-number/7IN57SP6LBLQZVDKRNMLDBN7U3/graph.json","events_json":"https://pith.science/api/pith-number/7IN57SP6LBLQZVDKRNMLDBN7U3/events.json","paper":"https://pith.science/paper/7IN57SP6"},"agent_actions":{"view_html":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3","download_json":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3.json","view_paper":"https://pith.science/paper/7IN57SP6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.20451&json=true","fetch_graph":"https://pith.science/api/pith-number/7IN57SP6LBLQZVDKRNMLDBN7U3/graph.json","fetch_events":"https://pith.science/api/pith-number/7IN57SP6LBLQZVDKRNMLDBN7U3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3/action/storage_attestation","attest_author":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3/action/author_attestation","sign_citation":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3/action/citation_signature","submit_replication":"https://pith.science/pith/7IN57SP6LBLQZVDKRNMLDBN7U3/action/replication_record"}},"created_at":"2026-07-05T11:46:20.692305+00:00","updated_at":"2026-07-05T11:46:20.692305+00:00"}