{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XD6WMHWWUZEAJXV5XPZC6YDQLS","short_pith_number":"pith:XD6WMHWW","schema_version":"1.0","canonical_sha256":"b8fd661ed6a64804debdbbf22f60705c9423c2ebe1980fa2e93a1ae9a0d975b8","source":{"kind":"arxiv","id":"2506.14794","version":1},"attestation_state":"computed","paper":{"title":"Assembly of Experts: Linear-time construction of the Chimera LLM variants with emergent and adaptable behaviors","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Benjamin Merkel, Daniel Klingmann, Dan Zecha, David A. Reiss, Fabian Klemm, Henrik Klagges, Robert Dahlke","submitted_at":"2025-05-31T18:23:19Z","abstract_excerpt":"Requiring $10^{13}$-$10^{15}$ FLOPs to calculate one 8 bit weight in an LLM during pretraining is extremely expensive and seems inefficient. To better leverage the huge investments made into pretrained models, we develop the new \"Assembly-of-Experts\" (AoE) construction method to create capable child variants of existing Mixture-of-Experts parent models in linear time. Model weight tensors get interpolated individually, allowing to enhance or suppress semantic features of the parents.\n  Varying the proportion of weights taken from the parent models, we observe some properties of the AoE child m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.14794","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-31T18:23:19Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a7d694f92a3a190fd2c77b4393ade2c3deb66995430759bd4337e820fb97242e","abstract_canon_sha256":"9762dafb09e2c128005a8bd416c4d2e426a0880578cfe9abfb7f8d9704ce7462"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:31.524861Z","signature_b64":"NJHtHPb5bmlmEN0/WTnh67ApSHMHZX+hvNpWwwka+tsTaliEAWqhUtMJr6y0syGen4+tfAWXv+nGctpsQQktBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8fd661ed6a64804debdbbf22f60705c9423c2ebe1980fa2e93a1ae9a0d975b8","last_reissued_at":"2026-07-05T11:23:31.524234Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:31.524234Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Assembly of Experts: Linear-time construction of the Chimera LLM variants with emergent and adaptable behaviors","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Benjamin Merkel, Daniel Klingmann, Dan Zecha, David A. Reiss, Fabian Klemm, Henrik Klagges, Robert Dahlke","submitted_at":"2025-05-31T18:23:19Z","abstract_excerpt":"Requiring $10^{13}$-$10^{15}$ FLOPs to calculate one 8 bit weight in an LLM during pretraining is extremely expensive and seems inefficient. To better leverage the huge investments made into pretrained models, we develop the new \"Assembly-of-Experts\" (AoE) construction method to create capable child variants of existing Mixture-of-Experts parent models in linear time. Model weight tensors get interpolated individually, allowing to enhance or suppress semantic features of the parents.\n  Varying the proportion of weights taken from the parent models, we observe some properties of the AoE child m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.14794","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.14794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.14794","created_at":"2026-07-05T11:23:31.524309+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.14794v1","created_at":"2026-07-05T11:23:31.524309+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.14794","created_at":"2026-07-05T11:23:31.524309+00:00"},{"alias_kind":"pith_short_12","alias_value":"XD6WMHWWUZEA","created_at":"2026-07-05T11:23:31.524309+00:00"},{"alias_kind":"pith_short_16","alias_value":"XD6WMHWWUZEAJXV5","created_at":"2026-07-05T11:23:31.524309+00:00"},{"alias_kind":"pith_short_8","alias_value":"XD6WMHWW","created_at":"2026-07-05T11:23:31.524309+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18098","citing_title":"IsabeLLM: Automated Theorem Proving Applied to Formally Verifying Consensus","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31308","citing_title":"Benchmarking Large Language Models on Floating-Point Error Classification","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":223,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS","json":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS.json","graph_json":"https://pith.science/api/pith-number/XD6WMHWWUZEAJXV5XPZC6YDQLS/graph.json","events_json":"https://pith.science/api/pith-number/XD6WMHWWUZEAJXV5XPZC6YDQLS/events.json","paper":"https://pith.science/paper/XD6WMHWW"},"agent_actions":{"view_html":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS","download_json":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS.json","view_paper":"https://pith.science/paper/XD6WMHWW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.14794&json=true","fetch_graph":"https://pith.science/api/pith-number/XD6WMHWWUZEAJXV5XPZC6YDQLS/graph.json","fetch_events":"https://pith.science/api/pith-number/XD6WMHWWUZEAJXV5XPZC6YDQLS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS/action/storage_attestation","attest_author":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS/action/author_attestation","sign_citation":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS/action/citation_signature","submit_replication":"https://pith.science/pith/XD6WMHWWUZEAJXV5XPZC6YDQLS/action/replication_record"}},"created_at":"2026-07-05T11:23:31.524309+00:00","updated_at":"2026-07-05T11:23:31.524309+00:00"}