{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:M3Y37W2J2AEF2P6JIFKFAIUA3Z","short_pith_number":"pith:M3Y37W2J","schema_version":"1.0","canonical_sha256":"66f1bfdb49d0085d3fc94154502280de422184544d2735ea702d810d10eb05e9","source":{"kind":"arxiv","id":"2408.01343","version":2},"attestation_state":"computed","paper":{"title":"StitchFusion: Weaving Any Visual Modalities to Enhance Multimodal Semantic Segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bingyu Li, Da Zhang, Junyu Gao, Xuelong Li, Zhiyuan Zhao","submitted_at":"2024-08-02T15:41:16Z","abstract_excerpt":"Multimodal semantic segmentation shows significant potential for enhancing segmentation accuracy in complex scenes. However, current methods often incorporate specialized feature fusion modules tailored to specific modalities, thereby restricting input flexibility and increasing the number of training parameters. To address these challenges, we propose StitchFusion, a straightforward yet effective modal fusion framework that integrates large-scale pre-trained models directly as encoders and feature fusers. This approach facilitates comprehensive multi-modal and multi-scale feature fusion, acco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.01343","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-02T15:41:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3983de68f56a7d01c636c3ff983af3562886df6425b214ba113d6bbe938990fe","abstract_canon_sha256":"a0035c65a065eea564894667e2580db5249cd65af51d33f77be58f935001b1b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:46.168557Z","signature_b64":"KUM/xTUaTSrWrH1/q+o0Cq8q6410lB3nrHBpjo6/puu2tJvYOvvYcbbVlQfSmkX1c5kY65qAMHUaVYWln8R2Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"66f1bfdb49d0085d3fc94154502280de422184544d2735ea702d810d10eb05e9","last_reissued_at":"2026-07-05T11:49:46.168019Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:46.168019Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StitchFusion: Weaving Any Visual Modalities to Enhance Multimodal Semantic Segmentation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Bingyu Li, Da Zhang, Junyu Gao, Xuelong Li, Zhiyuan Zhao","submitted_at":"2024-08-02T15:41:16Z","abstract_excerpt":"Multimodal semantic segmentation shows significant potential for enhancing segmentation accuracy in complex scenes. However, current methods often incorporate specialized feature fusion modules tailored to specific modalities, thereby restricting input flexibility and increasing the number of training parameters. To address these challenges, we propose StitchFusion, a straightforward yet effective modal fusion framework that integrates large-scale pre-trained models directly as encoders and feature fusers. This approach facilitates comprehensive multi-modal and multi-scale feature fusion, acco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.01343","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.01343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.01343","created_at":"2026-07-05T11:49:46.168076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.01343v2","created_at":"2026-07-05T11:49:46.168076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.01343","created_at":"2026-07-05T11:49:46.168076+00:00"},{"alias_kind":"pith_short_12","alias_value":"M3Y37W2J2AEF","created_at":"2026-07-05T11:49:46.168076+00:00"},{"alias_kind":"pith_short_16","alias_value":"M3Y37W2J2AEF2P6J","created_at":"2026-07-05T11:49:46.168076+00:00"},{"alias_kind":"pith_short_8","alias_value":"M3Y37W2J","created_at":"2026-07-05T11:49:46.168076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.15652","citing_title":"Towards Realistic Open-Vocabulary Remote Sensing Segmentation: Benchmark and Baseline","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z","json":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z.json","graph_json":"https://pith.science/api/pith-number/M3Y37W2J2AEF2P6JIFKFAIUA3Z/graph.json","events_json":"https://pith.science/api/pith-number/M3Y37W2J2AEF2P6JIFKFAIUA3Z/events.json","paper":"https://pith.science/paper/M3Y37W2J"},"agent_actions":{"view_html":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z","download_json":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z.json","view_paper":"https://pith.science/paper/M3Y37W2J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.01343&json=true","fetch_graph":"https://pith.science/api/pith-number/M3Y37W2J2AEF2P6JIFKFAIUA3Z/graph.json","fetch_events":"https://pith.science/api/pith-number/M3Y37W2J2AEF2P6JIFKFAIUA3Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z/action/storage_attestation","attest_author":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z/action/author_attestation","sign_citation":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z/action/citation_signature","submit_replication":"https://pith.science/pith/M3Y37W2J2AEF2P6JIFKFAIUA3Z/action/replication_record"}},"created_at":"2026-07-05T11:49:46.168076+00:00","updated_at":"2026-07-05T11:49:46.168076+00:00"}