{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:E5C2JAJJRG3IDF6T3GRNYHUQPC","short_pith_number":"pith:E5C2JAJJ","canonical_record":{"source":{"id":"1809.05848","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2018-09-16T10:17:37Z","cross_cats_sorted":[],"title_canon_sha256":"0e2834b2c7ab378f4f327f1dec507dfadff51e39ac8d17b7cc4957e0cae7fa46","abstract_canon_sha256":"01a8bbddbef3be32ce753082c279cdf0dd32b5bbff87ffe7c1f01cba36b49332"},"schema_version":"1.0"},"canonical_sha256":"2745a4812989b68197d3d9a2dc1e90789c2175d3bcf77f96e50f53777719299e","source":{"kind":"arxiv","id":"1809.05848","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1809.05848","created_at":"2026-05-18T00:04:35Z"},{"alias_kind":"arxiv_version","alias_value":"1809.05848v4","created_at":"2026-05-18T00:04:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1809.05848","created_at":"2026-05-18T00:04:35Z"},{"alias_kind":"pith_short_12","alias_value":"E5C2JAJJRG3I","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_16","alias_value":"E5C2JAJJRG3IDF6T","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_8","alias_value":"E5C2JAJJ","created_at":"2026-05-18T12:32:19Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:E5C2JAJJRG3IDF6T3GRNYHUQPC","target":"record","payload":{"canonical_record":{"source":{"id":"1809.05848","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2018-09-16T10:17:37Z","cross_cats_sorted":[],"title_canon_sha256":"0e2834b2c7ab378f4f327f1dec507dfadff51e39ac8d17b7cc4957e0cae7fa46","abstract_canon_sha256":"01a8bbddbef3be32ce753082c279cdf0dd32b5bbff87ffe7c1f01cba36b49332"},"schema_version":"1.0"},"canonical_sha256":"2745a4812989b68197d3d9a2dc1e90789c2175d3bcf77f96e50f53777719299e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:04:35.390091Z","signature_b64":"3iXv+D8bMHdtg5L3f/zkuCcH9e3BJnqyd1sZR+uzg9oG4as0Qcc9zL6LI6YEKbz15TUW6pQ2+sLJRpO7hp2BAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2745a4812989b68197d3d9a2dc1e90789c2175d3bcf77f96e50f53777719299e","last_reissued_at":"2026-05-18T00:04:35.389669Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:04:35.389669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1809.05848","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:04:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QhF9uq+/zEeAp23ZHKcHHMDMW7Ofr0Gv7OBPyJHbdN55IsTTBUXlDNKbJzR1R0mQSbWugfOuZD7cNp/IAeO7Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-29T19:53:14.328856Z"},"content_sha256":"167efb9089fcc47e82061ffcd3e9df7e0ab387a490d4175324a6ec00707c77fd","schema_version":"1.0","event_id":"sha256:167efb9089fcc47e82061ffcd3e9df7e0ab387a490d4175324a6ec00707c77fd"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:E5C2JAJJRG3IDF6T3GRNYHUQPC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Towards Good Practices for Multi-modal Fusion in Large-scale Video Classification","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Changhu Wang, Jinlai Liu, Zehuan Yuan","submitted_at":"2018-09-16T10:17:37Z","abstract_excerpt":"Leveraging both visual frames and audio has been experimentally proven effective to improve large-scale video classification. Previous research on video classification mainly focuses on the analysis of visual content among extracted video frames and their temporal feature aggregation. In contrast, multimodal data fusion is achieved by simple operators like average and concatenation. Inspired by the success of bilinear pooling in the visual and language fusion, we introduce multi-modal factorized bilinear pooling (MFB) to fuse visual and audio representations. We combine MFB with different vide"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1809.05848","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:04:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Zj5JWSIgzMb2fIeywY9mRTplUZvShiiOE4rvsJ7uuBHQD12kyFjooz/PbU99ad0iBpEgv50jY7xAyquh2OrCCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-29T19:53:14.329529Z"},"content_sha256":"00e2bc89abd46941b66ac8a566a0061f5363605170566704eb40b408d94f2be3","schema_version":"1.0","event_id":"sha256:00e2bc89abd46941b66ac8a566a0061f5363605170566704eb40b408d94f2be3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC/bundle.json","state_url":"https://pith.science/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-29T19:53:14Z","links":{"resolver":"https://pith.science/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC","bundle":"https://pith.science/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC/bundle.json","state":"https://pith.science/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/E5C2JAJJRG3IDF6T3GRNYHUQPC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:E5C2JAJJRG3IDF6T3GRNYHUQPC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"01a8bbddbef3be32ce753082c279cdf0dd32b5bbff87ffe7c1f01cba36b49332","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2018-09-16T10:17:37Z","title_canon_sha256":"0e2834b2c7ab378f4f327f1dec507dfadff51e39ac8d17b7cc4957e0cae7fa46"},"schema_version":"1.0","source":{"id":"1809.05848","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1809.05848","created_at":"2026-05-18T00:04:35Z"},{"alias_kind":"arxiv_version","alias_value":"1809.05848v4","created_at":"2026-05-18T00:04:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1809.05848","created_at":"2026-05-18T00:04:35Z"},{"alias_kind":"pith_short_12","alias_value":"E5C2JAJJRG3I","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_16","alias_value":"E5C2JAJJRG3IDF6T","created_at":"2026-05-18T12:32:19Z"},{"alias_kind":"pith_short_8","alias_value":"E5C2JAJJ","created_at":"2026-05-18T12:32:19Z"}],"graph_snapshots":[{"event_id":"sha256:00e2bc89abd46941b66ac8a566a0061f5363605170566704eb40b408d94f2be3","target":"graph","created_at":"2026-05-18T00:04:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Leveraging both visual frames and audio has been experimentally proven effective to improve large-scale video classification. Previous research on video classification mainly focuses on the analysis of visual content among extracted video frames and their temporal feature aggregation. In contrast, multimodal data fusion is achieved by simple operators like average and concatenation. Inspired by the success of bilinear pooling in the visual and language fusion, we introduce multi-modal factorized bilinear pooling (MFB) to fuse visual and audio representations. We combine MFB with different vide","authors_text":"Changhu Wang, Jinlai Liu, Zehuan Yuan","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2018-09-16T10:17:37Z","title":"Towards Good Practices for Multi-modal Fusion in Large-scale Video Classification"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1809.05848","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:167efb9089fcc47e82061ffcd3e9df7e0ab387a490d4175324a6ec00707c77fd","target":"record","created_at":"2026-05-18T00:04:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"01a8bbddbef3be32ce753082c279cdf0dd32b5bbff87ffe7c1f01cba36b49332","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2018-09-16T10:17:37Z","title_canon_sha256":"0e2834b2c7ab378f4f327f1dec507dfadff51e39ac8d17b7cc4957e0cae7fa46"},"schema_version":"1.0","source":{"id":"1809.05848","kind":"arxiv","version":4}},"canonical_sha256":"2745a4812989b68197d3d9a2dc1e90789c2175d3bcf77f96e50f53777719299e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2745a4812989b68197d3d9a2dc1e90789c2175d3bcf77f96e50f53777719299e","first_computed_at":"2026-05-18T00:04:35.389669Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:04:35.389669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"3iXv+D8bMHdtg5L3f/zkuCcH9e3BJnqyd1sZR+uzg9oG4as0Qcc9zL6LI6YEKbz15TUW6pQ2+sLJRpO7hp2BAg==","signature_status":"signed_v1","signed_at":"2026-05-18T00:04:35.390091Z","signed_message":"canonical_sha256_bytes"},"source_id":"1809.05848","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:167efb9089fcc47e82061ffcd3e9df7e0ab387a490d4175324a6ec00707c77fd","sha256:00e2bc89abd46941b66ac8a566a0061f5363605170566704eb40b408d94f2be3"],"state_sha256":"7c8603dabd17e4d6a5c37596cf326357f25c15ef9040e377af4f33ace06b115a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ktqrhfdY+5cxxXkHHMfvNu6elZIS0JZeUewy7iQEQsSL7P53ifzE+Fli578ew2whQ1o4oKgLs8J+7+kpwlYSAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-29T19:53:14.333157Z","bundle_sha256":"db129c4085a1618d3f8bb32e2fc23ad23a4d2138490063bfa2e89cfbc586c9b4"}}