{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LDKFFU6UMRIQH6TITGOJYIMJL7","short_pith_number":"pith:LDKFFU6U","schema_version":"1.0","canonical_sha256":"58d452d3d4645103fa68999c9c21895ff4466afa4b0ec1aa2966d354c2827a18","source":{"kind":"arxiv","id":"2412.12628","version":2},"attestation_state":"computed","paper":{"title":"Dense Audio-Visual Event Localization under Cross-Modal Consistency and Multi-Temporal Granularity Collaboration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dan Guo, Jinxing Zhou, Shengeng Tang, Wei Qian, Xiaojun Chang, Ziheng Zhou","submitted_at":"2024-12-17T07:43:36Z","abstract_excerpt":"In the field of audio-visual learning, most research tasks focus exclusively on short videos. This paper focuses on the more practical Dense Audio-Visual Event Localization (DAVEL) task, advancing audio-visual scene understanding for longer, untrimmed videos. This task seeks to identify and temporally pinpoint all events simultaneously occurring in both audio and visual streams. Typically, each video encompasses dense events of multiple classes, which may overlap on the timeline, each exhibiting varied durations. Given these challenges, effectively exploiting the audio-visual relations and the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12628","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-17T07:43:36Z","cross_cats_sorted":[],"title_canon_sha256":"3e19279d7af93506454b6a51957766b919e80169d137699f66bebceb15dab67d","abstract_canon_sha256":"c22a98a33ab1a545765dba0bd084ce11e9027b985e9d947c491dfdaf21f647e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:57.261890Z","signature_b64":"ng8H8c1SYPZ5hrftbTOtMrA7cqZ3MwjZ+bVsHHaeKrzkmV+85lOHVq9ezzu+iaT/sq+2Sz8EcwxJQ3vMLRRABQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58d452d3d4645103fa68999c9c21895ff4466afa4b0ec1aa2966d354c2827a18","last_reissued_at":"2026-07-05T09:50:57.261198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:57.261198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dense Audio-Visual Event Localization under Cross-Modal Consistency and Multi-Temporal Granularity Collaboration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dan Guo, Jinxing Zhou, Shengeng Tang, Wei Qian, Xiaojun Chang, Ziheng Zhou","submitted_at":"2024-12-17T07:43:36Z","abstract_excerpt":"In the field of audio-visual learning, most research tasks focus exclusively on short videos. This paper focuses on the more practical Dense Audio-Visual Event Localization (DAVEL) task, advancing audio-visual scene understanding for longer, untrimmed videos. This task seeks to identify and temporally pinpoint all events simultaneously occurring in both audio and visual streams. Typically, each video encompasses dense events of multiple classes, which may overlap on the timeline, each exhibiting varied durations. Given these challenges, effectively exploiting the audio-visual relations and the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12628","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12628","created_at":"2026-07-05T09:50:57.261269+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12628v2","created_at":"2026-07-05T09:50:57.261269+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12628","created_at":"2026-07-05T09:50:57.261269+00:00"},{"alias_kind":"pith_short_12","alias_value":"LDKFFU6UMRIQ","created_at":"2026-07-05T09:50:57.261269+00:00"},{"alias_kind":"pith_short_16","alias_value":"LDKFFU6UMRIQH6TI","created_at":"2026-07-05T09:50:57.261269+00:00"},{"alias_kind":"pith_short_8","alias_value":"LDKFFU6U","created_at":"2026-07-05T09:50:57.261269+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7","json":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7.json","graph_json":"https://pith.science/api/pith-number/LDKFFU6UMRIQH6TITGOJYIMJL7/graph.json","events_json":"https://pith.science/api/pith-number/LDKFFU6UMRIQH6TITGOJYIMJL7/events.json","paper":"https://pith.science/paper/LDKFFU6U"},"agent_actions":{"view_html":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7","download_json":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7.json","view_paper":"https://pith.science/paper/LDKFFU6U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12628&json=true","fetch_graph":"https://pith.science/api/pith-number/LDKFFU6UMRIQH6TITGOJYIMJL7/graph.json","fetch_events":"https://pith.science/api/pith-number/LDKFFU6UMRIQH6TITGOJYIMJL7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7/action/storage_attestation","attest_author":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7/action/author_attestation","sign_citation":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7/action/citation_signature","submit_replication":"https://pith.science/pith/LDKFFU6UMRIQH6TITGOJYIMJL7/action/replication_record"}},"created_at":"2026-07-05T09:50:57.261269+00:00","updated_at":"2026-07-05T09:50:57.261269+00:00"}