{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:FW25LWG5I57JGZ3IIF24GZ2BQS","short_pith_number":"pith:FW25LWG5","schema_version":"1.0","canonical_sha256":"2db5d5d8dd477e9367684175c36741848529fc565621854f08f9459ffa250683","source":{"kind":"arxiv","id":"1909.10838","version":2},"attestation_state":"computed","paper":{"title":"Talk2Car: Taking Control of Your Self-Driving Car","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.RO"],"primary_cat":"cs.AI","authors_text":"Dusan Grujicic, Luc Van Gool, Marie-Francine Moens, Simon Vandenhende, Thierry Deruyttere","submitted_at":"2019-09-24T12:29:27Z","abstract_excerpt":"A long-term goal of artificial intelligence is to have an agent execute commands communicated through natural language. In many cases the commands are grounded in a visual environment shared by the human who gives the command and the agent. Execution of the command then requires mapping the command into the physical visual space, after which the appropriate action can be taken. In this paper we consider the former. Or more specifically, we consider the problem in an autonomous driving setting, where a passenger requests an action that can be associated with an object found in a street scene. O"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.10838","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2019-09-24T12:29:27Z","cross_cats_sorted":["cs.CL","cs.RO"],"title_canon_sha256":"6b3f3664c81080fe64ed0d0a42e6f9a9577d021bd6408a03b55acc144c3cd9a0","abstract_canon_sha256":"3d1f612031971c236294c7608329669d04c801c29131b164232ee70007989146"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:30:38.809884Z","signature_b64":"F9bMx1Uk9I761T/XbDk3gUlsFnqsBULiPrwuo4z3LsVhxzFAJTSBeIqXlEHkuiL9ZM3C9muQ1JXRK4dP/LZPAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2db5d5d8dd477e9367684175c36741848529fc565621854f08f9459ffa250683","last_reissued_at":"2026-07-05T01:30:38.809473Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:30:38.809473Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Talk2Car: Taking Control of Your Self-Driving Car","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.RO"],"primary_cat":"cs.AI","authors_text":"Dusan Grujicic, Luc Van Gool, Marie-Francine Moens, Simon Vandenhende, Thierry Deruyttere","submitted_at":"2019-09-24T12:29:27Z","abstract_excerpt":"A long-term goal of artificial intelligence is to have an agent execute commands communicated through natural language. In many cases the commands are grounded in a visual environment shared by the human who gives the command and the agent. Execution of the command then requires mapping the command into the physical visual space, after which the appropriate action can be taken. In this paper we consider the former. Or more specifically, we consider the problem in an autonomous driving setting, where a passenger requests an action that can be associated with an object found in a street scene. O"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.10838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.10838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.10838","created_at":"2026-07-05T01:30:38.809529+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.10838v2","created_at":"2026-07-05T01:30:38.809529+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.10838","created_at":"2026-07-05T01:30:38.809529+00:00"},{"alias_kind":"pith_short_12","alias_value":"FW25LWG5I57J","created_at":"2026-07-05T01:30:38.809529+00:00"},{"alias_kind":"pith_short_16","alias_value":"FW25LWG5I57JGZ3I","created_at":"2026-07-05T01:30:38.809529+00:00"},{"alias_kind":"pith_short_8","alias_value":"FW25LWG5","created_at":"2026-07-05T01:30:38.809529+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"1903.11027","citing_title":"nuScenes: A multimodal dataset for autonomous driving","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03454","citing_title":"Think Before You Drive: World Model-Inspired Multimodal Grounding for Autonomous Vehicles","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2410.22313","citing_title":"Senna: Bridging Large Vision-Language Models and End-to-End Autonomous Driving","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08052","citing_title":"ReCogDrive: A Reinforced Cognitive Framework for End-to-End Autonomous Driving","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2402.12289","citing_title":"DriveVLM: The Convergence of Autonomous Driving and Large Vision-Language Models","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS","json":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS.json","graph_json":"https://pith.science/api/pith-number/FW25LWG5I57JGZ3IIF24GZ2BQS/graph.json","events_json":"https://pith.science/api/pith-number/FW25LWG5I57JGZ3IIF24GZ2BQS/events.json","paper":"https://pith.science/paper/FW25LWG5"},"agent_actions":{"view_html":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS","download_json":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS.json","view_paper":"https://pith.science/paper/FW25LWG5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.10838&json=true","fetch_graph":"https://pith.science/api/pith-number/FW25LWG5I57JGZ3IIF24GZ2BQS/graph.json","fetch_events":"https://pith.science/api/pith-number/FW25LWG5I57JGZ3IIF24GZ2BQS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS/action/storage_attestation","attest_author":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS/action/author_attestation","sign_citation":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS/action/citation_signature","submit_replication":"https://pith.science/pith/FW25LWG5I57JGZ3IIF24GZ2BQS/action/replication_record"}},"created_at":"2026-07-05T01:30:38.809529+00:00","updated_at":"2026-07-05T01:30:38.809529+00:00"}