{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RMPZYYQEATYWJ2UUZ7MJ2PND5Z","short_pith_number":"pith:RMPZYYQE","schema_version":"1.0","canonical_sha256":"8b1f9c620404f164ea94cfd89d3da3ee57f3faf3f15bdde0aba528b23ee30d59","source":{"kind":"arxiv","id":"2505.08725","version":1},"attestation_state":"computed","paper":{"title":"Extending Large Vision-Language Model for Diverse Interactive Tasks in Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bing Wang, Dingkang Liang, Dingyuan Zhang, Haoyu Fu, Hongwei Xie, Xiang Bai, Xin Zhou, Zongchuang Zhao","submitted_at":"2025-05-13T16:36:51Z","abstract_excerpt":"The Large Visual-Language Models (LVLMs) have significantly advanced image understanding. Their comprehension and reasoning capabilities enable promising applications in autonomous driving scenarios. However, existing research typically focuses on front-view perspectives and partial objects within scenes, struggling to achieve comprehensive scene understanding. Meanwhile, existing LVLMs suffer from the lack of mapping relationship between 2D and 3D and insufficient integration of 3D object localization and instruction understanding. To tackle these limitations, we first introduce NuInteract, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08725","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-13T16:36:51Z","cross_cats_sorted":[],"title_canon_sha256":"4b8dd804ba18fb277f85d3fc847da62b7d614c55b57f9aba27a6779c33772494","abstract_canon_sha256":"1310dc20d01ec9aff8a7f3ce912b1de90124d570b1781e4778f283e96c932b91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:35.368060Z","signature_b64":"/amB1ysp/KpctLYqtlHJgyl0vrK44TmqkWUT6SZ75t+lViOXZwhU5Zc6/+UpTP039NCkaBmjMFtT0s+8OgRbCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b1f9c620404f164ea94cfd89d3da3ee57f3faf3f15bdde0aba528b23ee30d59","last_reissued_at":"2026-07-05T11:02:35.367587Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:35.367587Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Extending Large Vision-Language Model for Diverse Interactive Tasks in Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bing Wang, Dingkang Liang, Dingyuan Zhang, Haoyu Fu, Hongwei Xie, Xiang Bai, Xin Zhou, Zongchuang Zhao","submitted_at":"2025-05-13T16:36:51Z","abstract_excerpt":"The Large Visual-Language Models (LVLMs) have significantly advanced image understanding. Their comprehension and reasoning capabilities enable promising applications in autonomous driving scenarios. However, existing research typically focuses on front-view perspectives and partial objects within scenes, struggling to achieve comprehensive scene understanding. Meanwhile, existing LVLMs suffer from the lack of mapping relationship between 2D and 3D and insufficient integration of 3D object localization and instruction understanding. To tackle these limitations, we first introduce NuInteract, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08725","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08725/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08725","created_at":"2026-07-05T11:02:35.367644+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08725v1","created_at":"2026-07-05T11:02:35.367644+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08725","created_at":"2026-07-05T11:02:35.367644+00:00"},{"alias_kind":"pith_short_12","alias_value":"RMPZYYQEATYW","created_at":"2026-07-05T11:02:35.367644+00:00"},{"alias_kind":"pith_short_16","alias_value":"RMPZYYQEATYWJ2UU","created_at":"2026-07-05T11:02:35.367644+00:00"},{"alias_kind":"pith_short_8","alias_value":"RMPZYYQE","created_at":"2026-07-05T11:02:35.367644+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.23180","citing_title":"GaussianDWM: 3D Gaussian Driving World Model for Unified Scene Understanding and Multi-Modal Generation","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28196","citing_title":"HERMES++: Toward a Unified Driving World Model for 3D Scene Understanding and Generation","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z","json":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z.json","graph_json":"https://pith.science/api/pith-number/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/graph.json","events_json":"https://pith.science/api/pith-number/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/events.json","paper":"https://pith.science/paper/RMPZYYQE"},"agent_actions":{"view_html":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z","download_json":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z.json","view_paper":"https://pith.science/paper/RMPZYYQE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08725&json=true","fetch_graph":"https://pith.science/api/pith-number/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/graph.json","fetch_events":"https://pith.science/api/pith-number/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/action/storage_attestation","attest_author":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/action/author_attestation","sign_citation":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/action/citation_signature","submit_replication":"https://pith.science/pith/RMPZYYQEATYWJ2UUZ7MJ2PND5Z/action/replication_record"}},"created_at":"2026-07-05T11:02:35.367644+00:00","updated_at":"2026-07-05T11:02:35.367644+00:00"}