{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FV3PIZXLCN24AFJ5BTI2UEP336","short_pith_number":"pith:FV3PIZXL","schema_version":"1.0","canonical_sha256":"2d76f466eb1375c0153d0cd1aa11fbdf84c9abd2242db6bdadaea6d5e2f45ae8","source":{"kind":"arxiv","id":"2311.09179","version":1},"attestation_state":"computed","paper":{"title":"SiRA: Sparse Mixture of Low Rank Adaptation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Canoee Liu, Chu-Cheng Lin, Han Lu, Jindong Chen, Lei Meng, Lei Shu, Liangchen Luo, Nevan Wichers, Tianlong Chen, Xinyi Wang, Yun Zhu","submitted_at":"2023-11-15T18:15:37Z","abstract_excerpt":"Parameter Efficient Tuning has been an prominent approach to adapt the Large Language Model to downstream tasks. Most previous works considers adding the dense trainable parameters, where all parameters are used to adapt certain task. We found this less effective empirically using the example of LoRA that introducing more trainable parameters does not help. Motivated by this we investigate the importance of leveraging \"sparse\" computation and propose SiRA: sparse mixture of low rank adaption. SiRA leverages the Sparse Mixture of Expert(SMoE) to boost the performance of LoRA. Specifically it en"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.09179","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-15T18:15:37Z","cross_cats_sorted":[],"title_canon_sha256":"b11d553c2143f08c652af63604c7e8afab8f095f56683bf91c9a3e1779fac105","abstract_canon_sha256":"4ba1c73b53498304ce0b6e00211fdb4f9ee6d01b2316e19f5d98df5957b2cac9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:13:09.952516Z","signature_b64":"SvCgEfL2MQmPjBMlKZTYmsKkrD2mNCoTVgAbudtydzKvYmTmmG7gJLtgnXnMRbrSaqrhjRJEl3lwFsbeFBjzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d76f466eb1375c0153d0cd1aa11fbdf84c9abd2242db6bdadaea6d5e2f45ae8","last_reissued_at":"2026-07-05T07:13:09.952095Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:13:09.952095Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SiRA: Sparse Mixture of Low Rank Adaptation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Canoee Liu, Chu-Cheng Lin, Han Lu, Jindong Chen, Lei Meng, Lei Shu, Liangchen Luo, Nevan Wichers, Tianlong Chen, Xinyi Wang, Yun Zhu","submitted_at":"2023-11-15T18:15:37Z","abstract_excerpt":"Parameter Efficient Tuning has been an prominent approach to adapt the Large Language Model to downstream tasks. Most previous works considers adding the dense trainable parameters, where all parameters are used to adapt certain task. We found this less effective empirically using the example of LoRA that introducing more trainable parameters does not help. Motivated by this we investigate the importance of leveraging \"sparse\" computation and propose SiRA: sparse mixture of low rank adaption. SiRA leverages the Sparse Mixture of Expert(SMoE) to boost the performance of LoRA. Specifically it en"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09179","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09179/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.09179","created_at":"2026-07-05T07:13:09.952152+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.09179v1","created_at":"2026-07-05T07:13:09.952152+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09179","created_at":"2026-07-05T07:13:09.952152+00:00"},{"alias_kind":"pith_short_12","alias_value":"FV3PIZXLCN24","created_at":"2026-07-05T07:13:09.952152+00:00"},{"alias_kind":"pith_short_16","alias_value":"FV3PIZXLCN24AFJ5","created_at":"2026-07-05T07:13:09.952152+00:00"},{"alias_kind":"pith_short_8","alias_value":"FV3PIZXL","created_at":"2026-07-05T07:13:09.952152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07111","citing_title":"Beyond LoRA vs. Full Fine-Tuning: Gradient-Guided Optimizer Routing for LLM Adaptation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07111","citing_title":"Beyond LoRA vs. Full Fine-Tuning: Gradient-Guided Optimizer Routing for LLM Adaptation","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336","json":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336.json","graph_json":"https://pith.science/api/pith-number/FV3PIZXLCN24AFJ5BTI2UEP336/graph.json","events_json":"https://pith.science/api/pith-number/FV3PIZXLCN24AFJ5BTI2UEP336/events.json","paper":"https://pith.science/paper/FV3PIZXL"},"agent_actions":{"view_html":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336","download_json":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336.json","view_paper":"https://pith.science/paper/FV3PIZXL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.09179&json=true","fetch_graph":"https://pith.science/api/pith-number/FV3PIZXLCN24AFJ5BTI2UEP336/graph.json","fetch_events":"https://pith.science/api/pith-number/FV3PIZXLCN24AFJ5BTI2UEP336/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336/action/storage_attestation","attest_author":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336/action/author_attestation","sign_citation":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336/action/citation_signature","submit_replication":"https://pith.science/pith/FV3PIZXLCN24AFJ5BTI2UEP336/action/replication_record"}},"created_at":"2026-07-05T07:13:09.952152+00:00","updated_at":"2026-07-05T07:13:09.952152+00:00"}