{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H4AS6ZNFMGNKTYHYVECIT45R4L","short_pith_number":"pith:H4AS6ZNF","schema_version":"1.0","canonical_sha256":"3f012f65a5619aa9e0f8a90489f3b1e2e8e30731737880268d377cacc015d6ab","source":{"kind":"arxiv","id":"2405.17730","version":1},"attestation_state":"computed","paper":{"title":"MMPareto: Boosting Multimodal Learning with Innocent Unimodal Assistance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Di Hu, Yake Wei","submitted_at":"2024-05-28T01:19:13Z","abstract_excerpt":"Multimodal learning methods with targeted unimodal learning objectives have exhibited their superior efficacy in alleviating the imbalanced multimodal learning problem. However, in this paper, we identify the previously ignored gradient conflict between multimodal and unimodal learning objectives, potentially misleading the unimodal encoder optimization. To well diminish these conflicts, we observe the discrepancy between multimodal loss and unimodal loss, where both gradient magnitude and covariance of the easier-to-learn multimodal loss are smaller than the unimodal one. With this property, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17730","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-28T01:19:13Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM"],"title_canon_sha256":"952f5802a880f8a76e559615ade06a1a2543394cec328f4767750b30dd9d8a01","abstract_canon_sha256":"e36889ce195bad72d3565bdcd9f9d3332af9c09562f4c18b99c41ee4aba4e3b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:10.231123Z","signature_b64":"ac1IYUiAkbRgMyb6JIvFe4KU4FkZhXble7Gs1sxgaDFUW0HC5qDc2r/NGPuqvjTuXjM/KJh7XG9tNGxV/noPCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f012f65a5619aa9e0f8a90489f3b1e2e8e30731737880268d377cacc015d6ab","last_reissued_at":"2026-07-05T08:24:10.230697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:10.230697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMPareto: Boosting Multimodal Learning with Innocent Unimodal Assistance","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Di Hu, Yake Wei","submitted_at":"2024-05-28T01:19:13Z","abstract_excerpt":"Multimodal learning methods with targeted unimodal learning objectives have exhibited their superior efficacy in alleviating the imbalanced multimodal learning problem. However, in this paper, we identify the previously ignored gradient conflict between multimodal and unimodal learning objectives, potentially misleading the unimodal encoder optimization. To well diminish these conflicts, we observe the discrepancy between multimodal loss and unimodal loss, where both gradient magnitude and covariance of the easier-to-learn multimodal loss are smaller than the unimodal one. With this property, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17730","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17730/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17730","created_at":"2026-07-05T08:24:10.230759+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17730v1","created_at":"2026-07-05T08:24:10.230759+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17730","created_at":"2026-07-05T08:24:10.230759+00:00"},{"alias_kind":"pith_short_12","alias_value":"H4AS6ZNFMGNK","created_at":"2026-07-05T08:24:10.230759+00:00"},{"alias_kind":"pith_short_16","alias_value":"H4AS6ZNFMGNKTYHY","created_at":"2026-07-05T08:24:10.230759+00:00"},{"alias_kind":"pith_short_8","alias_value":"H4AS6ZNF","created_at":"2026-07-05T08:24:10.230759+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17296","citing_title":"Pareto LoRA: Mitigating Modality Imbalance in Unified Multimodal Models via Pareto-Optimal Gradient Integration","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01856","citing_title":"Boosting Multimodal Federated Learning via Chained Modality Optimization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21670","citing_title":"Diverse via bounded Agreement: Geometric Regularization for Multimodal Fusion","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01833","citing_title":"Language-Pretraining-Induced Bias: A Strong Foundation for General Vision Tasks","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05773","citing_title":"PDMP: Rethinking Balanced Multimodal Learning via Performance-Dominant Modality Prioritization","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L","json":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L.json","graph_json":"https://pith.science/api/pith-number/H4AS6ZNFMGNKTYHYVECIT45R4L/graph.json","events_json":"https://pith.science/api/pith-number/H4AS6ZNFMGNKTYHYVECIT45R4L/events.json","paper":"https://pith.science/paper/H4AS6ZNF"},"agent_actions":{"view_html":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L","download_json":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L.json","view_paper":"https://pith.science/paper/H4AS6ZNF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17730&json=true","fetch_graph":"https://pith.science/api/pith-number/H4AS6ZNFMGNKTYHYVECIT45R4L/graph.json","fetch_events":"https://pith.science/api/pith-number/H4AS6ZNFMGNKTYHYVECIT45R4L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L/action/storage_attestation","attest_author":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L/action/author_attestation","sign_citation":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L/action/citation_signature","submit_replication":"https://pith.science/pith/H4AS6ZNFMGNKTYHYVECIT45R4L/action/replication_record"}},"created_at":"2026-07-05T08:24:10.230759+00:00","updated_at":"2026-07-05T08:24:10.230759+00:00"}