{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PWDXUCZNQLOMTJTCO6CPOZGUYD","short_pith_number":"pith:PWDXUCZN","schema_version":"1.0","canonical_sha256":"7d877a0b2d82dcc9a6627784f764d4c0dfd82f3159c6fa58bbec9c1c59f70e99","source":{"kind":"arxiv","id":"2312.00438","version":1},"attestation_state":"computed","paper":{"title":"Dolphins: Multimodal Language Model for Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaowei Xiao, Jiachen Sun, Marco Pavone, Yingzi Ma, Yulong Cao","submitted_at":"2023-12-01T09:10:33Z","abstract_excerpt":"The quest for fully autonomous vehicles (AVs) capable of navigating complex real-world scenarios with human-like understanding and responsiveness. In this paper, we introduce Dolphins, a novel vision-language model architected to imbibe human-like abilities as a conversational driving assistant. Dolphins is adept at processing multimodal inputs comprising video (or image) data, text instructions, and historical control signals to generate informed outputs corresponding to the provided instructions. Building upon the open-sourced pretrained Vision-Language Model, OpenFlamingo, we first enhance "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.00438","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-01T09:10:33Z","cross_cats_sorted":[],"title_canon_sha256":"02d05b74307c2eb0ebc30e5ff13fb199fd083d0ae42a98019e6e6b3921fc3e6f","abstract_canon_sha256":"a82a31db3d5c64409791fe5c866905626f55f03031f1144f474944d3e21bdcf6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:19:09.442834Z","signature_b64":"MbfDD4/20Hy8ywfebxehgvU6M+vbylcuzh0icDWLu85PU2S+kOqArzLKVL6q0OEkJCYuko6ELGUK7W79omJVCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d877a0b2d82dcc9a6627784f764d4c0dfd82f3159c6fa58bbec9c1c59f70e99","last_reissued_at":"2026-07-05T07:19:09.442336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:19:09.442336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dolphins: Multimodal Language Model for Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chaowei Xiao, Jiachen Sun, Marco Pavone, Yingzi Ma, Yulong Cao","submitted_at":"2023-12-01T09:10:33Z","abstract_excerpt":"The quest for fully autonomous vehicles (AVs) capable of navigating complex real-world scenarios with human-like understanding and responsiveness. In this paper, we introduce Dolphins, a novel vision-language model architected to imbibe human-like abilities as a conversational driving assistant. Dolphins is adept at processing multimodal inputs comprising video (or image) data, text instructions, and historical control signals to generate informed outputs corresponding to the provided instructions. Building upon the open-sourced pretrained Vision-Language Model, OpenFlamingo, we first enhance "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.00438","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.00438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.00438","created_at":"2026-07-05T07:19:09.442409+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.00438v1","created_at":"2026-07-05T07:19:09.442409+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.00438","created_at":"2026-07-05T07:19:09.442409+00:00"},{"alias_kind":"pith_short_12","alias_value":"PWDXUCZNQLOM","created_at":"2026-07-05T07:19:09.442409+00:00"},{"alias_kind":"pith_short_16","alias_value":"PWDXUCZNQLOMTJTC","created_at":"2026-07-05T07:19:09.442409+00:00"},{"alias_kind":"pith_short_8","alias_value":"PWDXUCZN","created_at":"2026-07-05T07:19:09.442409+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.18275","citing_title":"Visual Adversarial Attack on Vision-Language Models for Autonomous Driving","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD","json":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD.json","graph_json":"https://pith.science/api/pith-number/PWDXUCZNQLOMTJTCO6CPOZGUYD/graph.json","events_json":"https://pith.science/api/pith-number/PWDXUCZNQLOMTJTCO6CPOZGUYD/events.json","paper":"https://pith.science/paper/PWDXUCZN"},"agent_actions":{"view_html":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD","download_json":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD.json","view_paper":"https://pith.science/paper/PWDXUCZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.00438&json=true","fetch_graph":"https://pith.science/api/pith-number/PWDXUCZNQLOMTJTCO6CPOZGUYD/graph.json","fetch_events":"https://pith.science/api/pith-number/PWDXUCZNQLOMTJTCO6CPOZGUYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD/action/storage_attestation","attest_author":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD/action/author_attestation","sign_citation":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD/action/citation_signature","submit_replication":"https://pith.science/pith/PWDXUCZNQLOMTJTCO6CPOZGUYD/action/replication_record"}},"created_at":"2026-07-05T07:19:09.442409+00:00","updated_at":"2026-07-05T07:19:09.442409+00:00"}