{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ALKQC6IYLLJLF54KRQUEZB6O5W","short_pith_number":"pith:ALKQC6IY","schema_version":"1.0","canonical_sha256":"02d50179185ad2b2f78a8c284c87ceeda20d998709a6650de9397894199c2be5","source":{"kind":"arxiv","id":"2404.09275","version":1},"attestation_state":"computed","paper":{"title":"TrafficVLM: A Controllable Visual Language Model for Traffic Video Captioning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Anh Quan Dang, Hung Phong Tran, Minh Khoi Ho, Quang Minh Dinh","submitted_at":"2024-04-14T14:51:44Z","abstract_excerpt":"Traffic video description and analysis have received much attention recently due to the growing demand for efficient and reliable urban surveillance systems. Most existing methods only focus on locating traffic event segments, which severely lack descriptive details related to the behaviour and context of all the subjects of interest in the events. In this paper, we present TrafficVLM, a novel multi-modal dense video captioning model for vehicle ego camera view. TrafficVLM models traffic video events at different levels of analysis, both spatially and temporally, and generates long fine-graine"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.09275","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-04-14T14:51:44Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"d166c734d5ea434e60e1206eff1cecad5ef357e20be2551e956b4ab4b5165bb1","abstract_canon_sha256":"268e8d71ba6a54b9abcf84fc19b50a08ddbcc939a131ba3b47060d2ccf179462"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:39.449742Z","signature_b64":"c4+8xLdZkI4A6JDk+R8GNaz1O+mfiSxJ+fXUZT+7mfQvLGKC1MxGKxBM7fE9y6Djg2qEyRzxMBlkuaKHaZY/DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"02d50179185ad2b2f78a8c284c87ceeda20d998709a6650de9397894199c2be5","last_reissued_at":"2026-07-05T08:32:39.449269Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:39.449269Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TrafficVLM: A Controllable Visual Language Model for Traffic Video Captioning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Anh Quan Dang, Hung Phong Tran, Minh Khoi Ho, Quang Minh Dinh","submitted_at":"2024-04-14T14:51:44Z","abstract_excerpt":"Traffic video description and analysis have received much attention recently due to the growing demand for efficient and reliable urban surveillance systems. Most existing methods only focus on locating traffic event segments, which severely lack descriptive details related to the behaviour and context of all the subjects of interest in the events. In this paper, we present TrafficVLM, a novel multi-modal dense video captioning model for vehicle ego camera view. TrafficVLM models traffic video events at different levels of analysis, both spatially and temporally, and generates long fine-graine"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.09275","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.09275/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.09275","created_at":"2026-07-05T08:32:39.449333+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.09275v1","created_at":"2026-07-05T08:32:39.449333+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.09275","created_at":"2026-07-05T08:32:39.449333+00:00"},{"alias_kind":"pith_short_12","alias_value":"ALKQC6IYLLJL","created_at":"2026-07-05T08:32:39.449333+00:00"},{"alias_kind":"pith_short_16","alias_value":"ALKQC6IYLLJLF54K","created_at":"2026-07-05T08:32:39.449333+00:00"},{"alias_kind":"pith_short_8","alias_value":"ALKQC6IY","created_at":"2026-07-05T08:32:39.449333+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.02074","citing_title":"Large Language Models for Crash Detection in Video: A Survey of Methods, Datasets, and Challenges","ref_index":66,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W","json":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W.json","graph_json":"https://pith.science/api/pith-number/ALKQC6IYLLJLF54KRQUEZB6O5W/graph.json","events_json":"https://pith.science/api/pith-number/ALKQC6IYLLJLF54KRQUEZB6O5W/events.json","paper":"https://pith.science/paper/ALKQC6IY"},"agent_actions":{"view_html":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W","download_json":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W.json","view_paper":"https://pith.science/paper/ALKQC6IY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.09275&json=true","fetch_graph":"https://pith.science/api/pith-number/ALKQC6IYLLJLF54KRQUEZB6O5W/graph.json","fetch_events":"https://pith.science/api/pith-number/ALKQC6IYLLJLF54KRQUEZB6O5W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W/action/storage_attestation","attest_author":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W/action/author_attestation","sign_citation":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W/action/citation_signature","submit_replication":"https://pith.science/pith/ALKQC6IYLLJLF54KRQUEZB6O5W/action/replication_record"}},"created_at":"2026-07-05T08:32:39.449333+00:00","updated_at":"2026-07-05T08:32:39.449333+00:00"}