{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X7HHW2XVGLCP7Z5WSKYBWCBAZ2","short_pith_number":"pith:X7HHW2XV","schema_version":"1.0","canonical_sha256":"bfce7b6af532c4ffe7b692b01b0820ceb73c41898741a2bb70e746a0513ce7f4","source":{"kind":"arxiv","id":"2410.14972","version":3},"attestation_state":"computed","paper":{"title":"MENTOR: Mixture-of-Experts Network with Task-Oriented Perturbation for Visual Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Chenhao Lu, Guowei Xu, Huazhe Xu, Suning Huang, Tianhai Liang, Yihan Xu, Zhehao Kou, Zhengrong Xue, Zheyu Zhang","submitted_at":"2024-10-19T04:31:54Z","abstract_excerpt":"Visual deep reinforcement learning (RL) enables robots to acquire skills from visual input for unstructured tasks. However, current algorithms suffer from low sample efficiency, limiting their practical applicability. In this work, we present MENTOR, a method that improves both the architecture and optimization of RL agents. Specifically, MENTOR replaces the standard multi-layer perceptron (MLP) with a mixture-of-experts (MoE) backbone and introduces a task-oriented perturbation mechanism. MENTOR outperforms state-of-the-art methods across three simulation benchmarks and achieves an average of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.14972","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-10-19T04:31:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"512a85f268aca364cfdc56fa5f6383e046012313d5bbdbc66c026641fa2c4f20","abstract_canon_sha256":"8f2b745a071ece6fc45d47063d9b989839b864d8994d2ab4bbfa857819cf3831"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:40.660017Z","signature_b64":"/DCsJ4THc//2RUPRn+RUxCScSEj65KP2ZxjzkxvDOXVzhr3+z22SmM+7ZDrw/8QiS77tGAvNvZ5M+3cpaiZ6AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bfce7b6af532c4ffe7b692b01b0820ceb73c41898741a2bb70e746a0513ce7f4","last_reissued_at":"2026-07-05T11:31:40.659394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:40.659394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MENTOR: Mixture-of-Experts Network with Task-Oriented Perturbation for Visual Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Chenhao Lu, Guowei Xu, Huazhe Xu, Suning Huang, Tianhai Liang, Yihan Xu, Zhehao Kou, Zhengrong Xue, Zheyu Zhang","submitted_at":"2024-10-19T04:31:54Z","abstract_excerpt":"Visual deep reinforcement learning (RL) enables robots to acquire skills from visual input for unstructured tasks. However, current algorithms suffer from low sample efficiency, limiting their practical applicability. In this work, we present MENTOR, a method that improves both the architecture and optimization of RL agents. Specifically, MENTOR replaces the standard multi-layer perceptron (MLP) with a mixture-of-experts (MoE) backbone and introduces a task-oriented perturbation mechanism. MENTOR outperforms state-of-the-art methods across three simulation benchmarks and achieves an average of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.14972","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.14972/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.14972","created_at":"2026-07-05T11:31:40.659466+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.14972v3","created_at":"2026-07-05T11:31:40.659466+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.14972","created_at":"2026-07-05T11:31:40.659466+00:00"},{"alias_kind":"pith_short_12","alias_value":"X7HHW2XVGLCP","created_at":"2026-07-05T11:31:40.659466+00:00"},{"alias_kind":"pith_short_16","alias_value":"X7HHW2XVGLCP7Z5W","created_at":"2026-07-05T11:31:40.659466+00:00"},{"alias_kind":"pith_short_8","alias_value":"X7HHW2XV","created_at":"2026-07-05T11:31:40.659466+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10305","citing_title":"SARM2: Multi-Task Stage Aware Reward Modeling for Self Improving Robotic Manipulation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16278","citing_title":"DriveMoE: Mixture-of-Experts for Vision-Language-Action Model in End-to-End Autonomous Driving","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14009","citing_title":"GRaD-Nav++: Vision-Language Model Enabled Visual Drone Navigation with Gaussian Radiance Fields and Differentiable Dynamics","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17486","citing_title":"DyGRO-VLA: Cross-Task Scaling of Vision-Language-Action Models via Dynamic Grouped Residual Optimization","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23121","citing_title":"Breaking Lock-In: Preserving Steerability under Low-Data VLA Post-Training","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2","json":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2.json","graph_json":"https://pith.science/api/pith-number/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/graph.json","events_json":"https://pith.science/api/pith-number/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/events.json","paper":"https://pith.science/paper/X7HHW2XV"},"agent_actions":{"view_html":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2","download_json":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2.json","view_paper":"https://pith.science/paper/X7HHW2XV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.14972&json=true","fetch_graph":"https://pith.science/api/pith-number/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/graph.json","fetch_events":"https://pith.science/api/pith-number/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/action/storage_attestation","attest_author":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/action/author_attestation","sign_citation":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/action/citation_signature","submit_replication":"https://pith.science/pith/X7HHW2XVGLCP7Z5WSKYBWCBAZ2/action/replication_record"}},"created_at":"2026-07-05T11:31:40.659466+00:00","updated_at":"2026-07-05T11:31:40.659466+00:00"}