{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PHBSNYMRSCN7NLMRHLV4WOXKRY","short_pith_number":"pith:PHBSNYMR","schema_version":"1.0","canonical_sha256":"79c326e191909bf6ad913aebcb3aea8e008c588eb5c81408510e44e41b3c80b8","source":{"kind":"arxiv","id":"2407.19719","version":3},"attestation_state":"computed","paper":{"title":"Urban Safety Perception Assessments via Integrating Multimodal Large Language Models with Street View Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Wang, Jiaxin Zhang, Tomohiro Fukuda, Yunqin Li","submitted_at":"2024-07-29T06:03:13Z","abstract_excerpt":"Measuring urban safety perception is an important and complex task that traditionally relies heavily on human resources. This process often involves extensive field surveys, manual data collection, and subjective assessments, which can be time-consuming, costly, and sometimes inconsistent. Street View Images (SVIs), along with deep learning methods, provide a way to realize large-scale urban safety detection. However, achieving this goal often requires extensive human annotation to train safety ranking models, and the architectural differences between cities hinder the transferability of these"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.19719","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-29T06:03:13Z","cross_cats_sorted":[],"title_canon_sha256":"b84bba1eb4f00ffc064fbe75b7d96a0baa735ed95d9078d9d3e6a24ded560f5e","abstract_canon_sha256":"27a57027678ef14ec03b2ca638b22995878751ed86ab064b2665fa24b15fe51c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:57.096172Z","signature_b64":"x7AD04cUWstHK0+Zg0IwgDwRXTmc4yn80a3Lithn2E7qANKUnOaDYNaihQ7l1NPECOktqI2nqiecGJFQwXWwAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79c326e191909bf6ad913aebcb3aea8e008c588eb5c81408510e44e41b3c80b8","last_reissued_at":"2026-07-05T11:13:57.095588Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:57.095588Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Urban Safety Perception Assessments via Integrating Multimodal Large Language Models with Street View Images","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Wang, Jiaxin Zhang, Tomohiro Fukuda, Yunqin Li","submitted_at":"2024-07-29T06:03:13Z","abstract_excerpt":"Measuring urban safety perception is an important and complex task that traditionally relies heavily on human resources. This process often involves extensive field surveys, manual data collection, and subjective assessments, which can be time-consuming, costly, and sometimes inconsistent. Street View Images (SVIs), along with deep learning methods, provide a way to realize large-scale urban safety detection. However, achieving this goal often requires extensive human annotation to train safety ranking models, and the architectural differences between cities hinder the transferability of these"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.19719","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.19719/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.19719","created_at":"2026-07-05T11:13:57.095663+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.19719v3","created_at":"2026-07-05T11:13:57.095663+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.19719","created_at":"2026-07-05T11:13:57.095663+00:00"},{"alias_kind":"pith_short_12","alias_value":"PHBSNYMRSCN7","created_at":"2026-07-05T11:13:57.095663+00:00"},{"alias_kind":"pith_short_16","alias_value":"PHBSNYMRSCN7NLMR","created_at":"2026-07-05T11:13:57.095663+00:00"},{"alias_kind":"pith_short_8","alias_value":"PHBSNYMR","created_at":"2026-07-05T11:13:57.095663+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.05087","citing_title":"Interpretable Multimodal Framework for Human-Centered Street Assessment: Integrating Visual-Language Models for Perceptual Urban Diagnostics","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY","json":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY.json","graph_json":"https://pith.science/api/pith-number/PHBSNYMRSCN7NLMRHLV4WOXKRY/graph.json","events_json":"https://pith.science/api/pith-number/PHBSNYMRSCN7NLMRHLV4WOXKRY/events.json","paper":"https://pith.science/paper/PHBSNYMR"},"agent_actions":{"view_html":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY","download_json":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY.json","view_paper":"https://pith.science/paper/PHBSNYMR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.19719&json=true","fetch_graph":"https://pith.science/api/pith-number/PHBSNYMRSCN7NLMRHLV4WOXKRY/graph.json","fetch_events":"https://pith.science/api/pith-number/PHBSNYMRSCN7NLMRHLV4WOXKRY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY/action/storage_attestation","attest_author":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY/action/author_attestation","sign_citation":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY/action/citation_signature","submit_replication":"https://pith.science/pith/PHBSNYMRSCN7NLMRHLV4WOXKRY/action/replication_record"}},"created_at":"2026-07-05T11:13:57.095663+00:00","updated_at":"2026-07-05T11:13:57.095663+00:00"}