{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NRAH2UE6NCTCNWZS2OVCUU24QV","short_pith_number":"pith:NRAH2UE6","schema_version":"1.0","canonical_sha256":"6c407d509e68a626db32d3aa2a535c85445c4a6b6525c842534f6fd0707017d6","source":{"kind":"arxiv","id":"2202.13914","version":2},"attestation_state":"computed","paper":{"title":"Combining Modular Skills in Multitask Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Alessandro Sordoni, Edoardo M. Ponti, Siva Reddy, Yoshua Bengio","submitted_at":"2022-02-28T16:07:19Z","abstract_excerpt":"A modular design encourages neural models to disentangle and recombine different facets of knowledge to generalise more systematically to new tasks. In this work, we assume that each task is associated with a subset of latent discrete skills from a (potentially small) inventory. In turn, skills correspond to parameter-efficient (sparse / low-rank) model parameterisations. By jointly learning these and a task-skill allocation matrix, the network for each task is instantiated as the average of the parameters of active skills. To favour non-trivial soft partitions of skills across tasks, we exper"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.13914","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-28T16:07:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"0929ab16c09011cacf6b55aa03fe7620a02b1901f4cc53576c98a55df8ad93a9","abstract_canon_sha256":"0c1376c00624396eebce179e6ac43e0248c88afb94c7b1e7f63367926a84f7fa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:01:12.836336Z","signature_b64":"ZMpTB+R28pbYbzccl6Rm/dITXqnKHdwKPCiFh6NhEt0YoFy4sOR7sCNTvUw4VCL7D1uceZ4iMw5uEZrR1ywQCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c407d509e68a626db32d3aa2a535c85445c4a6b6525c842534f6fd0707017d6","last_reissued_at":"2026-07-05T04:01:12.835839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:01:12.835839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Combining Modular Skills in Multitask Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Alessandro Sordoni, Edoardo M. Ponti, Siva Reddy, Yoshua Bengio","submitted_at":"2022-02-28T16:07:19Z","abstract_excerpt":"A modular design encourages neural models to disentangle and recombine different facets of knowledge to generalise more systematically to new tasks. In this work, we assume that each task is associated with a subset of latent discrete skills from a (potentially small) inventory. In turn, skills correspond to parameter-efficient (sparse / low-rank) model parameterisations. By jointly learning these and a task-skill allocation matrix, the network for each task is instantiated as the average of the parameters of active skills. To favour non-trivial soft partitions of skills across tasks, we exper"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.13914","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.13914/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.13914","created_at":"2026-07-05T04:01:12.835903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.13914v2","created_at":"2026-07-05T04:01:12.835903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.13914","created_at":"2026-07-05T04:01:12.835903+00:00"},{"alias_kind":"pith_short_12","alias_value":"NRAH2UE6NCTC","created_at":"2026-07-05T04:01:12.835903+00:00"},{"alias_kind":"pith_short_16","alias_value":"NRAH2UE6NCTCNWZS","created_at":"2026-07-05T04:01:12.835903+00:00"},{"alias_kind":"pith_short_8","alias_value":"NRAH2UE6","created_at":"2026-07-05T04:01:12.835903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19079","citing_title":"ARIADNE: Agnostic Routing for Inference-time Adapter DyNamic sElection","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2402.09353","citing_title":"DoRA: Weight-Decomposed Low-Rank Adaptation","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18866","citing_title":"HMR-Net: Hierarchical Modular Routing for Cross-Domain Object Detection in Aerial Images","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV","json":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV.json","graph_json":"https://pith.science/api/pith-number/NRAH2UE6NCTCNWZS2OVCUU24QV/graph.json","events_json":"https://pith.science/api/pith-number/NRAH2UE6NCTCNWZS2OVCUU24QV/events.json","paper":"https://pith.science/paper/NRAH2UE6"},"agent_actions":{"view_html":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV","download_json":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV.json","view_paper":"https://pith.science/paper/NRAH2UE6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.13914&json=true","fetch_graph":"https://pith.science/api/pith-number/NRAH2UE6NCTCNWZS2OVCUU24QV/graph.json","fetch_events":"https://pith.science/api/pith-number/NRAH2UE6NCTCNWZS2OVCUU24QV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV/action/storage_attestation","attest_author":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV/action/author_attestation","sign_citation":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV/action/citation_signature","submit_replication":"https://pith.science/pith/NRAH2UE6NCTCNWZS2OVCUU24QV/action/replication_record"}},"created_at":"2026-07-05T04:01:12.835903+00:00","updated_at":"2026-07-05T04:01:12.835903+00:00"}