{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KEO4V4B4SS6XB7ISRX7JT2HIQO","short_pith_number":"pith:KEO4V4B4","schema_version":"1.0","canonical_sha256":"511dcaf03c94bd70fd128dfe99e8e883b15e44507861065762457753ccbb261f","source":{"kind":"arxiv","id":"2412.13708","version":2},"attestation_state":"computed","paper":{"title":"JoVALE: Detecting Human Actions in Video Using Audiovisual and Language Contexts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jisong Kim, Jun Won Choi, Seok Hwan Lee, Soo Won Seo, Taein Son","submitted_at":"2024-12-18T10:51:31Z","abstract_excerpt":"Video Action Detection (VAD) entails localizing and categorizing action instances within videos, which inherently consist of diverse information sources such as audio, visual cues, and surrounding scene contexts. Leveraging this multi-modal information effectively for VAD poses a significant challenge, as the model must identify action-relevant cues with precision. In this study, we introduce a novel multi-modal VAD architecture, referred to as the Joint Actor-centric Visual, Audio, Language Encoder (JoVALE). JoVALE is the first VAD method to integrate audio and visual features with scene desc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.13708","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-18T10:51:31Z","cross_cats_sorted":[],"title_canon_sha256":"73dd902f395f609ae4cd0daf6a3828c5e7a449f29b74e30791e65d3456b66905","abstract_canon_sha256":"b09511548919215efb66628c969d21b2bccf1c25c79be3e9f5f42bcdccdfa5de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:26.336239Z","signature_b64":"DtcGMNUW9IDlnY5vbRZl3P3zJRwRyPrQpSa+92OOj6tIINsYFINkBCZnqhv1pRtheA1wgAA2RyXfzMy0yCcDAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"511dcaf03c94bd70fd128dfe99e8e883b15e44507861065762457753ccbb261f","last_reissued_at":"2026-07-05T10:08:26.335689Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:26.335689Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"JoVALE: Detecting Human Actions in Video Using Audiovisual and Language Contexts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jisong Kim, Jun Won Choi, Seok Hwan Lee, Soo Won Seo, Taein Son","submitted_at":"2024-12-18T10:51:31Z","abstract_excerpt":"Video Action Detection (VAD) entails localizing and categorizing action instances within videos, which inherently consist of diverse information sources such as audio, visual cues, and surrounding scene contexts. Leveraging this multi-modal information effectively for VAD poses a significant challenge, as the model must identify action-relevant cues with precision. In this study, we introduce a novel multi-modal VAD architecture, referred to as the Joint Actor-centric Visual, Audio, Language Encoder (JoVALE). JoVALE is the first VAD method to integrate audio and visual features with scene desc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.13708","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.13708/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.13708","created_at":"2026-07-05T10:08:26.335756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.13708v2","created_at":"2026-07-05T10:08:26.335756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.13708","created_at":"2026-07-05T10:08:26.335756+00:00"},{"alias_kind":"pith_short_12","alias_value":"KEO4V4B4SS6X","created_at":"2026-07-05T10:08:26.335756+00:00"},{"alias_kind":"pith_short_16","alias_value":"KEO4V4B4SS6XB7IS","created_at":"2026-07-05T10:08:26.335756+00:00"},{"alias_kind":"pith_short_8","alias_value":"KEO4V4B4","created_at":"2026-07-05T10:08:26.335756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO","json":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO.json","graph_json":"https://pith.science/api/pith-number/KEO4V4B4SS6XB7ISRX7JT2HIQO/graph.json","events_json":"https://pith.science/api/pith-number/KEO4V4B4SS6XB7ISRX7JT2HIQO/events.json","paper":"https://pith.science/paper/KEO4V4B4"},"agent_actions":{"view_html":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO","download_json":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO.json","view_paper":"https://pith.science/paper/KEO4V4B4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.13708&json=true","fetch_graph":"https://pith.science/api/pith-number/KEO4V4B4SS6XB7ISRX7JT2HIQO/graph.json","fetch_events":"https://pith.science/api/pith-number/KEO4V4B4SS6XB7ISRX7JT2HIQO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO/action/storage_attestation","attest_author":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO/action/author_attestation","sign_citation":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO/action/citation_signature","submit_replication":"https://pith.science/pith/KEO4V4B4SS6XB7ISRX7JT2HIQO/action/replication_record"}},"created_at":"2026-07-05T10:08:26.335756+00:00","updated_at":"2026-07-05T10:08:26.335756+00:00"}