{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5FZ5HJWTPVOVVHOH6BFXJNPURD","short_pith_number":"pith:5FZ5HJWT","schema_version":"1.0","canonical_sha256":"e973d3a6d37d5d5a9dc7f04b74b5f488cedce30f2c80760fc1efc2bcef5462c9","source":{"kind":"arxiv","id":"2502.02145","version":4},"attestation_state":"computed","paper":{"title":"From Words to Collisions: LLM-Guided Evaluation and Adversarial Generation of Safety-Critical Driving Scenarios","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.RO"],"primary_cat":"cs.AI","authors_text":"Amr Alanwar, Johannes Betz, Korbinian Moller, Mattia Piccinini, Yuan Gao","submitted_at":"2025-02-04T09:19:13Z","abstract_excerpt":"Ensuring the safety of autonomous vehicles requires virtual scenario-based testing, which depends on the robust evaluation and generation of safety-critical scenarios. So far, researchers have used scenario-based testing frameworks that rely heavily on handcrafted scenarios as safety metrics. To reduce the effort of human interpretation and overcome the limited scalability of these approaches, we combine Large Language Models (LLMs) with structured scenario parsing and prompt engineering to automatically evaluate and generate safety-critical driving scenarios. We introduce Cartesian and Ego-ce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02145","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-02-04T09:19:13Z","cross_cats_sorted":["cs.CL","cs.RO"],"title_canon_sha256":"28f3ceb120d2e148c0e84370606230659942bf68fa326ee8792d9f151533bb4c","abstract_canon_sha256":"b7d982235decedee00bc33b52a446ff45bd47156a382a255469e6fea2b5cd67b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:39:07.344264Z","signature_b64":"mygtBRBs+5qUJUfR4OhWzw9cpWnhWp9naRjgGV8P6WJo5VEByarpULJkpDAmDvJw1FcautWqmn1iO6rmlM9NCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e973d3a6d37d5d5a9dc7f04b74b5f488cedce30f2c80760fc1efc2bcef5462c9","last_reissued_at":"2026-07-05T11:39:07.343797Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:39:07.343797Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Words to Collisions: LLM-Guided Evaluation and Adversarial Generation of Safety-Critical Driving Scenarios","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.RO"],"primary_cat":"cs.AI","authors_text":"Amr Alanwar, Johannes Betz, Korbinian Moller, Mattia Piccinini, Yuan Gao","submitted_at":"2025-02-04T09:19:13Z","abstract_excerpt":"Ensuring the safety of autonomous vehicles requires virtual scenario-based testing, which depends on the robust evaluation and generation of safety-critical scenarios. So far, researchers have used scenario-based testing frameworks that rely heavily on handcrafted scenarios as safety metrics. To reduce the effort of human interpretation and overcome the limited scalability of these approaches, we combine Large Language Models (LLMs) with structured scenario parsing and prompt engineering to automatically evaluate and generate safety-critical driving scenarios. We introduce Cartesian and Ego-ce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02145","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02145","created_at":"2026-07-05T11:39:07.343854+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02145v4","created_at":"2026-07-05T11:39:07.343854+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02145","created_at":"2026-07-05T11:39:07.343854+00:00"},{"alias_kind":"pith_short_12","alias_value":"5FZ5HJWTPVOV","created_at":"2026-07-05T11:39:07.343854+00:00"},{"alias_kind":"pith_short_16","alias_value":"5FZ5HJWTPVOVVHOH","created_at":"2026-07-05T11:39:07.343854+00:00"},{"alias_kind":"pith_short_8","alias_value":"5FZ5HJWT","created_at":"2026-07-05T11:39:07.343854+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.25944","citing_title":"NuRisk: A Visual Question Answering Dataset for Agent-Level Risk Assessment in Autonomous Driving","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05929","citing_title":"LLM Harms: A Taxonomy and Discussion","ref_index":228,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD","json":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD.json","graph_json":"https://pith.science/api/pith-number/5FZ5HJWTPVOVVHOH6BFXJNPURD/graph.json","events_json":"https://pith.science/api/pith-number/5FZ5HJWTPVOVVHOH6BFXJNPURD/events.json","paper":"https://pith.science/paper/5FZ5HJWT"},"agent_actions":{"view_html":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD","download_json":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD.json","view_paper":"https://pith.science/paper/5FZ5HJWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02145&json=true","fetch_graph":"https://pith.science/api/pith-number/5FZ5HJWTPVOVVHOH6BFXJNPURD/graph.json","fetch_events":"https://pith.science/api/pith-number/5FZ5HJWTPVOVVHOH6BFXJNPURD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD/action/storage_attestation","attest_author":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD/action/author_attestation","sign_citation":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD/action/citation_signature","submit_replication":"https://pith.science/pith/5FZ5HJWTPVOVVHOH6BFXJNPURD/action/replication_record"}},"created_at":"2026-07-05T11:39:07.343854+00:00","updated_at":"2026-07-05T11:39:07.343854+00:00"}