{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:G4PKEYNAY3QOFQAQNXGN4TWB57","short_pith_number":"pith:G4PKEYNA","schema_version":"1.0","canonical_sha256":"371ea261a0c6e0e2c0106dccde4ec1eff1bd1219d82cd6af7ef9adf91ba6e725","source":{"kind":"arxiv","id":"2309.05186","version":2},"attestation_state":"computed","paper":{"title":"HiLM-D: Enhancing MLLMs with Multi-Scale High-Resolution Details for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Xu, Jianhua Han, Wei Zhang, Xiaomeng Li, Xinpeng Ding","submitted_at":"2023-09-11T01:24:13Z","abstract_excerpt":"Recent efforts to use natural language for interpretable driving focus mainly on planning, neglecting perception tasks. In this paper, we address this gap by introducing ROLISP (Risk Object Localization and Intention and Suggestion Prediction), which towards interpretable risk object detection and suggestion for ego car motions. Accurate ROLISP implementation requires extensive reasoning to identify critical traffic objects and infer their intentions, prompting us to explore the capabilities of multimodal large language models (MLLMs). However, the limited perception performance of CLIP-ViT vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.05186","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-11T01:24:13Z","cross_cats_sorted":[],"title_canon_sha256":"d3c25a70dea14038c9382de72726a4e08979a2b106b1ea4ada01ec12c057d84a","abstract_canon_sha256":"92cdfcc43d6326d551806007e3aa9b74d5b892a7d7e742226916f19e3d094b83"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:21.321272Z","signature_b64":"nDWRJOzp0Zp6V34DfJQqGtVEz5Cte+f9lymdVjfI2tMf+HaiDpZswSOFvHMf7PAt/z3xDUvg+frnJ1lG013XCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"371ea261a0c6e0e2c0106dccde4ec1eff1bd1219d82cd6af7ef9adf91ba6e725","last_reissued_at":"2026-07-05T10:37:21.320784Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:21.320784Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiLM-D: Enhancing MLLMs with Multi-Scale High-Resolution Details for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hang Xu, Jianhua Han, Wei Zhang, Xiaomeng Li, Xinpeng Ding","submitted_at":"2023-09-11T01:24:13Z","abstract_excerpt":"Recent efforts to use natural language for interpretable driving focus mainly on planning, neglecting perception tasks. In this paper, we address this gap by introducing ROLISP (Risk Object Localization and Intention and Suggestion Prediction), which towards interpretable risk object detection and suggestion for ego car motions. Accurate ROLISP implementation requires extensive reasoning to identify critical traffic objects and infer their intentions, prompting us to explore the capabilities of multimodal large language models (MLLMs). However, the limited perception performance of CLIP-ViT vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.05186","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.05186/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.05186","created_at":"2026-07-05T10:37:21.320840+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.05186v2","created_at":"2026-07-05T10:37:21.320840+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.05186","created_at":"2026-07-05T10:37:21.320840+00:00"},{"alias_kind":"pith_short_12","alias_value":"G4PKEYNAY3QO","created_at":"2026-07-05T10:37:21.320840+00:00"},{"alias_kind":"pith_short_16","alias_value":"G4PKEYNAY3QOFQAQ","created_at":"2026-07-05T10:37:21.320840+00:00"},{"alias_kind":"pith_short_8","alias_value":"G4PKEYNA","created_at":"2026-07-05T10:37:21.320840+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24759","citing_title":"UniDrive: A Unified Vision-Language and Grounding Framework for Interpretable Risk Understanding in Autonomous Driving","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13243","citing_title":"VADv2: End-to-End Vectorized Autonomous Driving via Probabilistic Planning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25944","citing_title":"NuRisk: A Visual Question Answering Dataset for Agent-Level Risk Assessment in Autonomous Driving","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04857","citing_title":"The Blind Spot of Adaptation: Quantifying and Mitigating Forgetting in Fine-tuned Driving Models","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57","json":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57.json","graph_json":"https://pith.science/api/pith-number/G4PKEYNAY3QOFQAQNXGN4TWB57/graph.json","events_json":"https://pith.science/api/pith-number/G4PKEYNAY3QOFQAQNXGN4TWB57/events.json","paper":"https://pith.science/paper/G4PKEYNA"},"agent_actions":{"view_html":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57","download_json":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57.json","view_paper":"https://pith.science/paper/G4PKEYNA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.05186&json=true","fetch_graph":"https://pith.science/api/pith-number/G4PKEYNAY3QOFQAQNXGN4TWB57/graph.json","fetch_events":"https://pith.science/api/pith-number/G4PKEYNAY3QOFQAQNXGN4TWB57/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57/action/storage_attestation","attest_author":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57/action/author_attestation","sign_citation":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57/action/citation_signature","submit_replication":"https://pith.science/pith/G4PKEYNAY3QOFQAQNXGN4TWB57/action/replication_record"}},"created_at":"2026-07-05T10:37:21.320840+00:00","updated_at":"2026-07-05T10:37:21.320840+00:00"}