{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HBJR64QU4UUSSCBPQSWYKCGUTV","short_pith_number":"pith:HBJR64QU","schema_version":"1.0","canonical_sha256":"38531f7214e52929082f84ad8508d49d54f6c59b41adfa526c965e7917f06215","source":{"kind":"arxiv","id":"2409.17519","version":1},"attestation_state":"computed","paper":{"title":"Robotic Environmental State Recognition with Pre-Trained Vision-Language Models and Black-Box Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Kei Okada, Kento Kawaharazuka, Masayuki Inaba, Naoaki Kanazawa, Yoshiki Obinata","submitted_at":"2024-09-26T04:02:20Z","abstract_excerpt":"In order for robots to autonomously navigate and operate in diverse environments, it is essential for them to recognize the state of their environment. On the other hand, the environmental state recognition has traditionally involved distinct methods tailored to each state to be recognized. In this study, we perform a unified environmental state recognition for robots through the spoken language with pre-trained large-scale vision-language models. We apply Visual Question Answering and Image-to-Text Retrieval, which are tasks of Vision-Language Models. We show that with our method, it is possi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.17519","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-09-26T04:02:20Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"deccf6f18b47d55ababc6056a0ccf9970aa0b54c2e9e1e6e7eb856eb4e1d0e67","abstract_canon_sha256":"56c0c3c0235135415dc4c4a6eaac06696e9c7d287b0b015149ef55d4f83927b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:11.322524Z","signature_b64":"U20/xUusNCshLGN4rfE6gMga/5RHgqdF6XcVdPaZTnYNAm3RfN16WnhJUkyBWXGGzdmEVRLpgmIpoarK4D5SDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"38531f7214e52929082f84ad8508d49d54f6c59b41adfa526c965e7917f06215","last_reissued_at":"2026-07-05T09:12:11.322004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:11.322004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robotic Environmental State Recognition with Pre-Trained Vision-Language Models and Black-Box Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.RO","authors_text":"Kei Okada, Kento Kawaharazuka, Masayuki Inaba, Naoaki Kanazawa, Yoshiki Obinata","submitted_at":"2024-09-26T04:02:20Z","abstract_excerpt":"In order for robots to autonomously navigate and operate in diverse environments, it is essential for them to recognize the state of their environment. On the other hand, the environmental state recognition has traditionally involved distinct methods tailored to each state to be recognized. In this study, we perform a unified environmental state recognition for robots through the spoken language with pre-trained large-scale vision-language models. We apply Visual Question Answering and Image-to-Text Retrieval, which are tasks of Vision-Language Models. We show that with our method, it is possi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.17519","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.17519/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.17519","created_at":"2026-07-05T09:12:11.322064+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.17519v1","created_at":"2026-07-05T09:12:11.322064+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.17519","created_at":"2026-07-05T09:12:11.322064+00:00"},{"alias_kind":"pith_short_12","alias_value":"HBJR64QU4UUS","created_at":"2026-07-05T09:12:11.322064+00:00"},{"alias_kind":"pith_short_16","alias_value":"HBJR64QU4UUSSCBP","created_at":"2026-07-05T09:12:11.322064+00:00"},{"alias_kind":"pith_short_8","alias_value":"HBJR64QU","created_at":"2026-07-05T09:12:11.322064+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV","json":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV.json","graph_json":"https://pith.science/api/pith-number/HBJR64QU4UUSSCBPQSWYKCGUTV/graph.json","events_json":"https://pith.science/api/pith-number/HBJR64QU4UUSSCBPQSWYKCGUTV/events.json","paper":"https://pith.science/paper/HBJR64QU"},"agent_actions":{"view_html":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV","download_json":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV.json","view_paper":"https://pith.science/paper/HBJR64QU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.17519&json=true","fetch_graph":"https://pith.science/api/pith-number/HBJR64QU4UUSSCBPQSWYKCGUTV/graph.json","fetch_events":"https://pith.science/api/pith-number/HBJR64QU4UUSSCBPQSWYKCGUTV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV/action/storage_attestation","attest_author":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV/action/author_attestation","sign_citation":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV/action/citation_signature","submit_replication":"https://pith.science/pith/HBJR64QU4UUSSCBPQSWYKCGUTV/action/replication_record"}},"created_at":"2026-07-05T09:12:11.322064+00:00","updated_at":"2026-07-05T09:12:11.322064+00:00"}