{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6HBIL7P2656WSCSPSCATNGAHMO","short_pith_number":"pith:6HBIL7P2","schema_version":"1.0","canonical_sha256":"f1c285fdfaf77d690a4f908136980763831ac254b3634ee4a656dd0542c54ff5","source":{"kind":"arxiv","id":"2306.08889","version":3},"attestation_state":"computed","paper":{"title":"Dissecting Multimodality in VideoQA Transformer Models by Impairing Modality Fusion","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Matyasko, Basura Fernando, Cheston Tan, Ishaan Singh Rawal, Shantanu Jaiswal","submitted_at":"2023-06-15T06:45:46Z","abstract_excerpt":"While VideoQA Transformer models demonstrate competitive performance on standard benchmarks, the reasons behind their success are not fully understood. Do these models capture the rich multimodal structures and dynamics from video and text jointly? Or are they achieving high scores by exploiting biases and spurious features? Hence, to provide insights, we design $\\textit{QUAG}$ (QUadrant AveraGe), a lightweight and non-parametric probe, to conduct dataset-model combined representation analysis by impairing modality fusion. We find that the models achieve high performance on many datasets witho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.08889","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-06-15T06:45:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b681c8fe9b48aa17e75a9f3490e01da6c888c17841504f2ebe813422a619f940","abstract_canon_sha256":"fb3bb0385f6c51da7e2ddb7a2e3b2de8d6a7191fb9e37506d290f17f0d3cf867"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:32.980136Z","signature_b64":"df1UrQu7c5W2D0SEodVzWIyIkWBu03KLbkC36tuOX9UKG/DlAL7PLTAxWaCxIC92p2xc6w/iPM//XErNB/1cBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1c285fdfaf77d690a4f908136980763831ac254b3634ee4a656dd0542c54ff5","last_reissued_at":"2026-07-05T08:28:32.979662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:32.979662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dissecting Multimodality in VideoQA Transformer Models by Impairing Modality Fusion","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Matyasko, Basura Fernando, Cheston Tan, Ishaan Singh Rawal, Shantanu Jaiswal","submitted_at":"2023-06-15T06:45:46Z","abstract_excerpt":"While VideoQA Transformer models demonstrate competitive performance on standard benchmarks, the reasons behind their success are not fully understood. Do these models capture the rich multimodal structures and dynamics from video and text jointly? Or are they achieving high scores by exploiting biases and spurious features? Hence, to provide insights, we design $\\textit{QUAG}$ (QUadrant AveraGe), a lightweight and non-parametric probe, to conduct dataset-model combined representation analysis by impairing modality fusion. We find that the models achieve high performance on many datasets witho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.08889","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.08889/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.08889","created_at":"2026-07-05T08:28:32.979719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.08889v3","created_at":"2026-07-05T08:28:32.979719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.08889","created_at":"2026-07-05T08:28:32.979719+00:00"},{"alias_kind":"pith_short_12","alias_value":"6HBIL7P2656W","created_at":"2026-07-05T08:28:32.979719+00:00"},{"alias_kind":"pith_short_16","alias_value":"6HBIL7P2656WSCSP","created_at":"2026-07-05T08:28:32.979719+00:00"},{"alias_kind":"pith_short_8","alias_value":"6HBIL7P2","created_at":"2026-07-05T08:28:32.979719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29915","citing_title":"H-GRPO: Permutation-Invariant Reinforcement Learning for Grounded Visual Reasoning","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO","json":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO.json","graph_json":"https://pith.science/api/pith-number/6HBIL7P2656WSCSPSCATNGAHMO/graph.json","events_json":"https://pith.science/api/pith-number/6HBIL7P2656WSCSPSCATNGAHMO/events.json","paper":"https://pith.science/paper/6HBIL7P2"},"agent_actions":{"view_html":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO","download_json":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO.json","view_paper":"https://pith.science/paper/6HBIL7P2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.08889&json=true","fetch_graph":"https://pith.science/api/pith-number/6HBIL7P2656WSCSPSCATNGAHMO/graph.json","fetch_events":"https://pith.science/api/pith-number/6HBIL7P2656WSCSPSCATNGAHMO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO/action/storage_attestation","attest_author":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO/action/author_attestation","sign_citation":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO/action/citation_signature","submit_replication":"https://pith.science/pith/6HBIL7P2656WSCSPSCATNGAHMO/action/replication_record"}},"created_at":"2026-07-05T08:28:32.979719+00:00","updated_at":"2026-07-05T08:28:32.979719+00:00"}