{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EZYZEGEFDQZEEGD57XO4F24SGW","short_pith_number":"pith:EZYZEGEF","schema_version":"1.0","canonical_sha256":"26719218851c3242187dfdddc2eb923583155f174d8c53c88cdf3b75605afbde","source":{"kind":"arxiv","id":"2405.07930","version":2},"attestation_state":"computed","paper":{"title":"Improving Multimodal Learning with Multi-Loss Gradient Modulation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Christos Chatzichristos, Konstantinos Kontras, Maarten De Vos, Matthew Blaschko","submitted_at":"2024-05-13T17:01:28Z","abstract_excerpt":"Learning from multiple modalities, such as audio and video, offers opportunities for leveraging complementary information, enhancing robustness, and improving contextual understanding and performance. However, combining such modalities presents challenges, especially when modalities differ in data structure, predictive contribution, and the complexity of their learning processes. It has been observed that one modality can potentially dominate the learning process, hindering the effective utilization of information from other modalities and leading to sub-optimal model performance. To address t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.07930","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.MM","submitted_at":"2024-05-13T17:01:28Z","cross_cats_sorted":["cs.CV","cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"fe7f3ad21e7fdfd852d312c378d76ddaa9567a3123fe5370f1f5300b75e48401","abstract_canon_sha256":"678a318804b576293ad176efbd766fcb7a68e2dc7a6e4957fca2f606c7aea038"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:57.282447Z","signature_b64":"MQMiPKnHyPqeRLejkWmWE9+eHuFHoqEI51PONUQ8vJ6UhWcMorzVXTvUPuOtFq5mc+UY5J6BV5WfgqxfQohmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26719218851c3242187dfdddc2eb923583155f174d8c53c88cdf3b75605afbde","last_reissued_at":"2026-07-05T09:19:57.281886Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:57.281886Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Multimodal Learning with Multi-Loss Gradient Modulation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"Christos Chatzichristos, Konstantinos Kontras, Maarten De Vos, Matthew Blaschko","submitted_at":"2024-05-13T17:01:28Z","abstract_excerpt":"Learning from multiple modalities, such as audio and video, offers opportunities for leveraging complementary information, enhancing robustness, and improving contextual understanding and performance. However, combining such modalities presents challenges, especially when modalities differ in data structure, predictive contribution, and the complexity of their learning processes. It has been observed that one modality can potentially dominate the learning process, hindering the effective utilization of information from other modalities and leading to sub-optimal model performance. To address t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.07930","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.07930/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.07930","created_at":"2026-07-05T09:19:57.281955+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.07930v2","created_at":"2026-07-05T09:19:57.281955+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.07930","created_at":"2026-07-05T09:19:57.281955+00:00"},{"alias_kind":"pith_short_12","alias_value":"EZYZEGEFDQZE","created_at":"2026-07-05T09:19:57.281955+00:00"},{"alias_kind":"pith_short_16","alias_value":"EZYZEGEFDQZEEGD5","created_at":"2026-07-05T09:19:57.281955+00:00"},{"alias_kind":"pith_short_8","alias_value":"EZYZEGEF","created_at":"2026-07-05T09:19:57.281955+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11614","citing_title":"Information-Theoretic Decomposition for Multimodal Interaction Learning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09853","citing_title":"SynIB: Informational Bottleneck for Maximizing Synergy in Multimodal Learning","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28869","citing_title":"Balancing Multimodal Learning through Label Space Reshaping","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16165","citing_title":"Second-Order Multi-Level Variance Correction for Modality Competition in Multimodal Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01833","citing_title":"Language-Pretraining-Induced Bias: A Strong Foundation for General Vision Tasks","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW","json":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW.json","graph_json":"https://pith.science/api/pith-number/EZYZEGEFDQZEEGD57XO4F24SGW/graph.json","events_json":"https://pith.science/api/pith-number/EZYZEGEFDQZEEGD57XO4F24SGW/events.json","paper":"https://pith.science/paper/EZYZEGEF"},"agent_actions":{"view_html":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW","download_json":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW.json","view_paper":"https://pith.science/paper/EZYZEGEF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.07930&json=true","fetch_graph":"https://pith.science/api/pith-number/EZYZEGEFDQZEEGD57XO4F24SGW/graph.json","fetch_events":"https://pith.science/api/pith-number/EZYZEGEFDQZEEGD57XO4F24SGW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW/action/storage_attestation","attest_author":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW/action/author_attestation","sign_citation":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW/action/citation_signature","submit_replication":"https://pith.science/pith/EZYZEGEFDQZEEGD57XO4F24SGW/action/replication_record"}},"created_at":"2026-07-05T09:19:57.281955+00:00","updated_at":"2026-07-05T09:19:57.281955+00:00"}