{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5NOCEGFGN5A5AW7GU7WMXPJQMB","short_pith_number":"pith:5NOCEGFG","schema_version":"1.0","canonical_sha256":"eb5c2218a66f41d05be6a7eccbbd30607e0eddb62dc2dad14a48669d8ea6a37d","source":{"kind":"arxiv","id":"2308.16493","version":1},"attestation_state":"computed","paper":{"title":"Expanding Frozen Vision-Language Models without Retraining: Towards Improved Robot Perception","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Mani Amani, Reza Akhavian, Riley Tavassoli","submitted_at":"2023-08-31T06:53:55Z","abstract_excerpt":"Vision-language models (VLMs) have shown powerful capabilities in visual question answering and reasoning tasks by combining visual representations with the abstract skill set large language models (LLMs) learn during pretraining. Vision, while the most popular modality to augment LLMs with, is only one representation of a scene. In human-robot interaction scenarios, robot perception requires accurate scene understanding by the robot. In this paper, we define and demonstrate a method of aligning the embedding spaces of different modalities (in this case, inertial measurement unit (IMU) data) t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.16493","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-08-31T06:53:55Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"ab1236f74611c3b815bbd7480a7d7489cf24373461774cd7722f50ef8ebc6976","abstract_canon_sha256":"442132935192c28e8059dc8eefec96688873083ae53eda5dfb20b0781384b77f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:46:28.431534Z","signature_b64":"AAO2cyk1FK3L+yUttekD6/3V5257jlnULNIVKruicjfN3z3YPxnYceLCWdjLU1LWb4MJbnB91Ni6rGlz0y0ABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb5c2218a66f41d05be6a7eccbbd30607e0eddb62dc2dad14a48669d8ea6a37d","last_reissued_at":"2026-07-05T06:46:28.431131Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:46:28.431131Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Expanding Frozen Vision-Language Models without Retraining: Towards Improved Robot Perception","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Mani Amani, Reza Akhavian, Riley Tavassoli","submitted_at":"2023-08-31T06:53:55Z","abstract_excerpt":"Vision-language models (VLMs) have shown powerful capabilities in visual question answering and reasoning tasks by combining visual representations with the abstract skill set large language models (LLMs) learn during pretraining. Vision, while the most popular modality to augment LLMs with, is only one representation of a scene. In human-robot interaction scenarios, robot perception requires accurate scene understanding by the robot. In this paper, we define and demonstrate a method of aligning the embedding spaces of different modalities (in this case, inertial measurement unit (IMU) data) t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.16493","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.16493/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.16493","created_at":"2026-07-05T06:46:28.431185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.16493v1","created_at":"2026-07-05T06:46:28.431185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.16493","created_at":"2026-07-05T06:46:28.431185+00:00"},{"alias_kind":"pith_short_12","alias_value":"5NOCEGFGN5A5","created_at":"2026-07-05T06:46:28.431185+00:00"},{"alias_kind":"pith_short_16","alias_value":"5NOCEGFGN5A5AW7G","created_at":"2026-07-05T06:46:28.431185+00:00"},{"alias_kind":"pith_short_8","alias_value":"5NOCEGFG","created_at":"2026-07-05T06:46:28.431185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10087","citing_title":"Foundation Model Driven Robotics: A Comprehensive Review","ref_index":71,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB","json":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB.json","graph_json":"https://pith.science/api/pith-number/5NOCEGFGN5A5AW7GU7WMXPJQMB/graph.json","events_json":"https://pith.science/api/pith-number/5NOCEGFGN5A5AW7GU7WMXPJQMB/events.json","paper":"https://pith.science/paper/5NOCEGFG"},"agent_actions":{"view_html":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB","download_json":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB.json","view_paper":"https://pith.science/paper/5NOCEGFG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.16493&json=true","fetch_graph":"https://pith.science/api/pith-number/5NOCEGFGN5A5AW7GU7WMXPJQMB/graph.json","fetch_events":"https://pith.science/api/pith-number/5NOCEGFGN5A5AW7GU7WMXPJQMB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB/action/storage_attestation","attest_author":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB/action/author_attestation","sign_citation":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB/action/citation_signature","submit_replication":"https://pith.science/pith/5NOCEGFGN5A5AW7GU7WMXPJQMB/action/replication_record"}},"created_at":"2026-07-05T06:46:28.431185+00:00","updated_at":"2026-07-05T06:46:28.431185+00:00"}