{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RJSKBPED2X5AIKWCIUK4S2ORXE","short_pith_number":"pith:RJSKBPED","schema_version":"1.0","canonical_sha256":"8a64a0bc83d5fa042ac24515c969d1b92ada08f40aff8b2f7f2aa9b72bffd1c9","source":{"kind":"arxiv","id":"2401.04210","version":1},"attestation_state":"computed","paper":{"title":"FunnyNet-W: Multimodal Learning of Funny Moments in Videos in the Wild","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Robin Courant, Vicky Kalogeiton, Zhi-Song Liu","submitted_at":"2024-01-08T19:39:36Z","abstract_excerpt":"Automatically understanding funny moments (i.e., the moments that make people laugh) when watching comedy is challenging, as they relate to various features, such as body language, dialogues and culture. In this paper, we propose FunnyNet-W, a model that relies on cross- and self-attention for visual, audio and text data to predict funny moments in videos. Unlike most methods that rely on ground truth data in the form of subtitles, in this work we exploit modalities that come naturally with videos: (a) video frames as they contain visual information indispensable for scene understanding, (b) a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.04210","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-08T19:39:36Z","cross_cats_sorted":["cs.AI","cs.CL","cs.MM","cs.SD","eess.AS"],"title_canon_sha256":"3b958cb099927965970bd24bdbd92c937e0fdd75fb48c01de4282dba4cf98ece","abstract_canon_sha256":"45bfac8312e58b4066b7325e2a8bf47bd706721deae221912371578964511dd5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:31:40.039517Z","signature_b64":"1BMlValLvkjmIWqChAC4GVOmpKRnKdh6nRspDp8RljsaWjycv0u91jPHC1YU4IkStp8zHfcvkWPoBND66pZgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a64a0bc83d5fa042ac24515c969d1b92ada08f40aff8b2f7f2aa9b72bffd1c9","last_reissued_at":"2026-07-05T07:31:40.039042Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:31:40.039042Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FunnyNet-W: Multimodal Learning of Funny Moments in Videos in the Wild","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.MM","cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Robin Courant, Vicky Kalogeiton, Zhi-Song Liu","submitted_at":"2024-01-08T19:39:36Z","abstract_excerpt":"Automatically understanding funny moments (i.e., the moments that make people laugh) when watching comedy is challenging, as they relate to various features, such as body language, dialogues and culture. In this paper, we propose FunnyNet-W, a model that relies on cross- and self-attention for visual, audio and text data to predict funny moments in videos. Unlike most methods that rely on ground truth data in the form of subtitles, in this work we exploit modalities that come naturally with videos: (a) video frames as they contain visual information indispensable for scene understanding, (b) a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.04210","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.04210/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.04210","created_at":"2026-07-05T07:31:40.039099+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.04210v1","created_at":"2026-07-05T07:31:40.039099+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.04210","created_at":"2026-07-05T07:31:40.039099+00:00"},{"alias_kind":"pith_short_12","alias_value":"RJSKBPED2X5A","created_at":"2026-07-05T07:31:40.039099+00:00"},{"alias_kind":"pith_short_16","alias_value":"RJSKBPED2X5AIKWC","created_at":"2026-07-05T07:31:40.039099+00:00"},{"alias_kind":"pith_short_8","alias_value":"RJSKBPED","created_at":"2026-07-05T07:31:40.039099+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00046","citing_title":"When Jokes Cross the Line: Analyzing Regular Humor and Dark Humor in YouTube Shorts","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE","json":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE.json","graph_json":"https://pith.science/api/pith-number/RJSKBPED2X5AIKWCIUK4S2ORXE/graph.json","events_json":"https://pith.science/api/pith-number/RJSKBPED2X5AIKWCIUK4S2ORXE/events.json","paper":"https://pith.science/paper/RJSKBPED"},"agent_actions":{"view_html":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE","download_json":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE.json","view_paper":"https://pith.science/paper/RJSKBPED","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.04210&json=true","fetch_graph":"https://pith.science/api/pith-number/RJSKBPED2X5AIKWCIUK4S2ORXE/graph.json","fetch_events":"https://pith.science/api/pith-number/RJSKBPED2X5AIKWCIUK4S2ORXE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE/action/storage_attestation","attest_author":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE/action/author_attestation","sign_citation":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE/action/citation_signature","submit_replication":"https://pith.science/pith/RJSKBPED2X5AIKWCIUK4S2ORXE/action/replication_record"}},"created_at":"2026-07-05T07:31:40.039099+00:00","updated_at":"2026-07-05T07:31:40.039099+00:00"}