{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:T53TOYAYICWUN6DWM5ULBHOT3M","short_pith_number":"pith:T53TOYAY","schema_version":"1.0","canonical_sha256":"9f7737601840ad46f8766768b09dd3db37ceb987a8c92542189a990517a1b042","source":{"kind":"arxiv","id":"2303.05674","version":2},"attestation_state":"computed","paper":{"title":"Robotic Applications of Pre-Trained Vision-Language Models to Various Recognition Behaviors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Kei Okada, Kento Kawaharazuka, Masayuki Inaba, Naoaki Kanazawa, Yoshiki Obinata","submitted_at":"2023-03-10T02:55:50Z","abstract_excerpt":"In recent years, a number of models that learn the relations between vision and language from large datasets have been released. These models perform a variety of tasks, such as answering questions about images, retrieving sentences that best correspond to images, and finding regions in images that correspond to phrases. Although there are some examples, the connection between these pre-trained vision-language models and robotics is still weak. If they are directly connected to robot motions, they lose their versatility due to the embodiment of the robot and the difficulty of data collection, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.05674","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-03-10T02:55:50Z","cross_cats_sorted":[],"title_canon_sha256":"230d92289a18f121749ca61a338d94922d548f3c929c59c5a080b923e006b61d","abstract_canon_sha256":"f52c49473bf35e3e5a20c3b6f261992447526783bebd36f0e793ee7cec177ce1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:47.093191Z","signature_b64":"JhmppwZQIznL1A3QJejqKMWm73PpMDU8da7GqIAZPXyGJyedAnGOx930znLNxH8QvyKB2ES1EeV9BMVw9A2AAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f7737601840ad46f8766768b09dd3db37ceb987a8c92542189a990517a1b042","last_reissued_at":"2026-07-05T07:56:47.092699Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:47.092699Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robotic Applications of Pre-Trained Vision-Language Models to Various Recognition Behaviors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Kei Okada, Kento Kawaharazuka, Masayuki Inaba, Naoaki Kanazawa, Yoshiki Obinata","submitted_at":"2023-03-10T02:55:50Z","abstract_excerpt":"In recent years, a number of models that learn the relations between vision and language from large datasets have been released. These models perform a variety of tasks, such as answering questions about images, retrieving sentences that best correspond to images, and finding regions in images that correspond to phrases. Although there are some examples, the connection between these pre-trained vision-language models and robotics is still weak. If they are directly connected to robot motions, they lose their versatility due to the embodiment of the robot and the difficulty of data collection, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.05674","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.05674/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.05674","created_at":"2026-07-05T07:56:47.092780+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.05674v2","created_at":"2026-07-05T07:56:47.092780+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.05674","created_at":"2026-07-05T07:56:47.092780+00:00"},{"alias_kind":"pith_short_12","alias_value":"T53TOYAYICWU","created_at":"2026-07-05T07:56:47.092780+00:00"},{"alias_kind":"pith_short_16","alias_value":"T53TOYAYICWUN6DW","created_at":"2026-07-05T07:56:47.092780+00:00"},{"alias_kind":"pith_short_8","alias_value":"T53TOYAY","created_at":"2026-07-05T07:56:47.092780+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M","json":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M.json","graph_json":"https://pith.science/api/pith-number/T53TOYAYICWUN6DWM5ULBHOT3M/graph.json","events_json":"https://pith.science/api/pith-number/T53TOYAYICWUN6DWM5ULBHOT3M/events.json","paper":"https://pith.science/paper/T53TOYAY"},"agent_actions":{"view_html":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M","download_json":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M.json","view_paper":"https://pith.science/paper/T53TOYAY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.05674&json=true","fetch_graph":"https://pith.science/api/pith-number/T53TOYAYICWUN6DWM5ULBHOT3M/graph.json","fetch_events":"https://pith.science/api/pith-number/T53TOYAYICWUN6DWM5ULBHOT3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M/action/storage_attestation","attest_author":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M/action/author_attestation","sign_citation":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M/action/citation_signature","submit_replication":"https://pith.science/pith/T53TOYAYICWUN6DWM5ULBHOT3M/action/replication_record"}},"created_at":"2026-07-05T07:56:47.092780+00:00","updated_at":"2026-07-05T07:56:47.092780+00:00"}