{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GGSMQB6X2CFH6YT57Z53FQWJ47","short_pith_number":"pith:GGSMQB6X","schema_version":"1.0","canonical_sha256":"31a4c807d7d08a7f627dfe7bb2c2c9e7d56439f12bd6b074a7c08c991709f455","source":{"kind":"arxiv","id":"2403.19838","version":2},"attestation_state":"computed","paper":{"title":"Multi-Frame, Lightweight & Efficient Vision-Language Models for Question Answering in Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Akshay Gopalkrishnan, Mohan Trivedi, Ross Greer","submitted_at":"2024-03-28T21:18:33Z","abstract_excerpt":"Vision-Language Models (VLMs) and Multi-Modal Language models (MMLMs) have become prominent in autonomous driving research, as these models can provide interpretable textual reasoning and responses for end-to-end autonomous driving safety tasks using traffic scene images and other data modalities. However, current approaches to these systems use expensive large language model (LLM) backbones and image encoders, making such systems unsuitable for real-time autonomous driving systems where tight memory constraints exist and fast inference time is necessary. To address these previous issues, we d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.19838","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-28T21:18:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ea7837781ee9d9aea9840ded1817c5da9b47d3f9a4e7c43077c3808814b482f8","abstract_canon_sha256":"bc4d8ed9c14a27d3ec92ef2ce8b96c457e4454bd90218fa4d22bb2a0fc52c728"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:17:14.044250Z","signature_b64":"7lHYKrHHKJnuhtGTLdRdi+8JZoKE9TTTlMRwmx0z29xlzvL5s6vZ4VjXMpkxOI+UFNgWb8PbnQBfLP4MgssNAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31a4c807d7d08a7f627dfe7bb2c2c9e7d56439f12bd6b074a7c08c991709f455","last_reissued_at":"2026-07-05T08:17:14.043639Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:17:14.043639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Frame, Lightweight & Efficient Vision-Language Models for Question Answering in Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Akshay Gopalkrishnan, Mohan Trivedi, Ross Greer","submitted_at":"2024-03-28T21:18:33Z","abstract_excerpt":"Vision-Language Models (VLMs) and Multi-Modal Language models (MMLMs) have become prominent in autonomous driving research, as these models can provide interpretable textual reasoning and responses for end-to-end autonomous driving safety tasks using traffic scene images and other data modalities. However, current approaches to these systems use expensive large language model (LLM) backbones and image encoders, making such systems unsuitable for real-time autonomous driving systems where tight memory constraints exist and fast inference time is necessary. To address these previous issues, we d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.19838","created_at":"2026-07-05T08:17:14.043783+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.19838v2","created_at":"2026-07-05T08:17:14.043783+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19838","created_at":"2026-07-05T08:17:14.043783+00:00"},{"alias_kind":"pith_short_12","alias_value":"GGSMQB6X2CFH","created_at":"2026-07-05T08:17:14.043783+00:00"},{"alias_kind":"pith_short_16","alias_value":"GGSMQB6X2CFH6YT5","created_at":"2026-07-05T08:17:14.043783+00:00"},{"alias_kind":"pith_short_8","alias_value":"GGSMQB6X","created_at":"2026-07-05T08:17:14.043783+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01518","citing_title":"Overthink-Triggered Slowdown Attacks on LVLM-Based Robotic Systems","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26038","citing_title":"DRScaffold: Boosting Dense-Scene Reasoning in Lightweight Vision Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04630","citing_title":"Multimodal Backdoor Attack on VLMs for Autonomous Driving via Graffiti and Cross-Lingual Triggers","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47","json":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47.json","graph_json":"https://pith.science/api/pith-number/GGSMQB6X2CFH6YT57Z53FQWJ47/graph.json","events_json":"https://pith.science/api/pith-number/GGSMQB6X2CFH6YT57Z53FQWJ47/events.json","paper":"https://pith.science/paper/GGSMQB6X"},"agent_actions":{"view_html":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47","download_json":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47.json","view_paper":"https://pith.science/paper/GGSMQB6X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.19838&json=true","fetch_graph":"https://pith.science/api/pith-number/GGSMQB6X2CFH6YT57Z53FQWJ47/graph.json","fetch_events":"https://pith.science/api/pith-number/GGSMQB6X2CFH6YT57Z53FQWJ47/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47/action/storage_attestation","attest_author":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47/action/author_attestation","sign_citation":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47/action/citation_signature","submit_replication":"https://pith.science/pith/GGSMQB6X2CFH6YT57Z53FQWJ47/action/replication_record"}},"created_at":"2026-07-05T08:17:14.043783+00:00","updated_at":"2026-07-05T08:17:14.043783+00:00"}