{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2MNO2QYR4PNFXVNZMYJ7XC6Y25","short_pith_number":"pith:2MNO2QYR","schema_version":"1.0","canonical_sha256":"d31aed4311e3da5bd5b96613fb8bd8d75c451a7693a6e051270725da2b01c7da","source":{"kind":"arxiv","id":"2505.18586","version":1},"attestation_state":"computed","paper":{"title":"Guiding the Experts: Semantic Priors for Efficient and Focused MoE Routing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengxi Min, Enver Sangineto, Qi Wang, Wei Wang, Weixin Ye, Yahui Liu, Yao Zhao","submitted_at":"2025-05-24T08:25:50Z","abstract_excerpt":"Mixture-of-Experts (MoE) models have emerged as a promising direction for scaling vision architectures efficiently. Among them, Soft MoE improves training stability by assigning each token to all experts via continuous dispatch weights. However, current designs overlook the semantic structure which is implicitly encoded in these weights, resulting in suboptimal expert routing. In this paper, we discover that dispatch weights in Soft MoE inherently exhibit segmentation-like patterns but are not explicitly aligned with semantic regions. Motivated by this observation, we propose a foreground-guid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18586","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-24T08:25:50Z","cross_cats_sorted":[],"title_canon_sha256":"83f90d5bf7e33da3b36a0ab6f93a83203ebca7ba8ae8229f92eb4cdf138492b0","abstract_canon_sha256":"acc081c1e431391116896887eb85c71a0c595a8b130060a17bb7cea03a1e6bc9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:16.083057Z","signature_b64":"n/mEhW+Cls4/NwRdg1toyhOAH8lPgkfS+DhKDzKptk+6qyBlqaLy1BzaMUH96XNlRS7sl6fPIGIalqSHfXBqDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d31aed4311e3da5bd5b96613fb8bd8d75c451a7693a6e051270725da2b01c7da","last_reissued_at":"2026-07-05T11:09:16.082571Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:16.082571Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guiding the Experts: Semantic Priors for Efficient and Focused MoE Routing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengxi Min, Enver Sangineto, Qi Wang, Wei Wang, Weixin Ye, Yahui Liu, Yao Zhao","submitted_at":"2025-05-24T08:25:50Z","abstract_excerpt":"Mixture-of-Experts (MoE) models have emerged as a promising direction for scaling vision architectures efficiently. Among them, Soft MoE improves training stability by assigning each token to all experts via continuous dispatch weights. However, current designs overlook the semantic structure which is implicitly encoded in these weights, resulting in suboptimal expert routing. In this paper, we discover that dispatch weights in Soft MoE inherently exhibit segmentation-like patterns but are not explicitly aligned with semantic regions. Motivated by this observation, we propose a foreground-guid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18586","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18586/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18586","created_at":"2026-07-05T11:09:16.082630+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18586v1","created_at":"2026-07-05T11:09:16.082630+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18586","created_at":"2026-07-05T11:09:16.082630+00:00"},{"alias_kind":"pith_short_12","alias_value":"2MNO2QYR4PNF","created_at":"2026-07-05T11:09:16.082630+00:00"},{"alias_kind":"pith_short_16","alias_value":"2MNO2QYR4PNFXVNZ","created_at":"2026-07-05T11:09:16.082630+00:00"},{"alias_kind":"pith_short_8","alias_value":"2MNO2QYR","created_at":"2026-07-05T11:09:16.082630+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20544","citing_title":"Toward Calibrated Mixture-of-Experts Under Distribution Shift","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25166","citing_title":"AME-TS: Anchored Mixture-of-Experts for Time Series Forecasting","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25","json":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25.json","graph_json":"https://pith.science/api/pith-number/2MNO2QYR4PNFXVNZMYJ7XC6Y25/graph.json","events_json":"https://pith.science/api/pith-number/2MNO2QYR4PNFXVNZMYJ7XC6Y25/events.json","paper":"https://pith.science/paper/2MNO2QYR"},"agent_actions":{"view_html":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25","download_json":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25.json","view_paper":"https://pith.science/paper/2MNO2QYR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18586&json=true","fetch_graph":"https://pith.science/api/pith-number/2MNO2QYR4PNFXVNZMYJ7XC6Y25/graph.json","fetch_events":"https://pith.science/api/pith-number/2MNO2QYR4PNFXVNZMYJ7XC6Y25/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25/action/storage_attestation","attest_author":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25/action/author_attestation","sign_citation":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25/action/citation_signature","submit_replication":"https://pith.science/pith/2MNO2QYR4PNFXVNZMYJ7XC6Y25/action/replication_record"}},"created_at":"2026-07-05T11:09:16.082630+00:00","updated_at":"2026-07-05T11:09:16.082630+00:00"}