{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:65PALSYVPD7Q6GMTXKZI7PB2OY","short_pith_number":"pith:65PALSYV","schema_version":"1.0","canonical_sha256":"f75e05cb1578ff0f1993bab28fbc3a7612518f740638b6a8b291df66c67a32bb","source":{"kind":"arxiv","id":"2602.12026","version":2},"attestation_state":"computed","paper":{"title":"Protein Circuit Tracing via Cross-layer Transcoders","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"ProtoMech applies cross-layer transcoders to protein language models to recover 82-89% of model performance using sparse circuits that match biological motifs and improve protein design in over 70% of cases.","cross_cats":["q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Amirali Aghazadeh, Daniel Saeedi, Darin Tsui, Kunal Talreja","submitted_at":"2026-02-12T14:57:57Z","abstract_excerpt":"Protein language models (pLMs) have emerged as powerful predictors of protein structure and function. However, the computational circuits underlying their predictions remain poorly understood. Recent mechanistic interpretability methods decompose pLM representations into interpretable features, but they treat each layer independently and thus fail to capture cross-layer computation, limiting their ability to approximate the full model. We introduce ProtoMech, a framework for discovering computational circuits in pLMs using cross-layer transcoders that learn sparse latent representations jointl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":true},"canonical_record":{"source":{"id":"2602.12026","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-12T14:57:57Z","cross_cats_sorted":["q-bio.QM"],"title_canon_sha256":"8986ec24dc1c664190b4d340f60ae0240eaf4cebea73a7b7250faaa7de5c7dec","abstract_canon_sha256":"a2da0fa6107cab3e2935b382d332a6affcbd602a471fa13b37e10674a68fd474"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T03:09:23.609361Z","signature_b64":"s2k4BIvi1E69VbHyqTAQg5OhF1PTs9YjZriOQHh91DgKTet1JlQBSU6lu6nbiU0d1CGBT9dtaQJntnRSm4QxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f75e05cb1578ff0f1993bab28fbc3a7612518f740638b6a8b291df66c67a32bb","last_reissued_at":"2026-05-18T03:09:23.608707Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T03:09:23.608707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Protein Circuit Tracing via Cross-layer Transcoders","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"ProtoMech applies cross-layer transcoders to protein language models to recover 82-89% of model performance using sparse circuits that match biological motifs and improve protein design in over 70% of cases.","cross_cats":["q-bio.QM"],"primary_cat":"cs.LG","authors_text":"Amirali Aghazadeh, Daniel Saeedi, Darin Tsui, Kunal Talreja","submitted_at":"2026-02-12T14:57:57Z","abstract_excerpt":"Protein language models (pLMs) have emerged as powerful predictors of protein structure and function. However, the computational circuits underlying their predictions remain poorly understood. Recent mechanistic interpretability methods decompose pLM representations into interpretable features, but they treat each layer independently and thus fail to capture cross-layer computation, limiting their ability to approximate the full model. We introduce ProtoMech, a framework for discovering computational circuits in pLMs using cross-layer transcoders that learn sparse latent representations jointl"},"claims":{"count":3,"items":[{"kind":"strongest_claim","text":"ProtoMech recovers 82-89% of the original performance on protein family classification and function prediction tasks. ProtoMech then identifies compressed circuits that use <1% of the latent space while retaining up to 79% of model accuracy... Steering along these circuits enables high-fitness protein design, surpassing baseline methods in more than 70% of cases.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the sparse latent representations learned jointly across layers faithfully approximate the model's full computational circuitry and that the identified circuits correspond to genuine structural and functional motifs rather than artifacts of the transcoder training.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"ProtoMech applies cross-layer transcoders to protein language models to recover 82-89% of model performance using sparse circuits that match biological motifs and improve protein design in over 70% of cases.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"}],"snapshot_sha256":"bed94d8513c72060ced23bce25a73c51ba5f066107f78e430cda4da1641d165d"},"source":{"id":"2602.12026","kind":"arxiv","version":2},"verdict":{"id":"1dabf5dd-5c46-4cac-9434-9ed2062519f7","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-16T01:49:55.793406Z","strongest_claim":"ProtoMech recovers 82-89% of the original performance on protein family classification and function prediction tasks. ProtoMech then identifies compressed circuits that use <1% of the latent space while retaining up to 79% of model accuracy... Steering along these circuits enables high-fitness protein design, surpassing baseline methods in more than 70% of cases.","one_line_summary":"ProtoMech applies cross-layer transcoders to protein language models to recover 82-89% of model performance using sparse circuits that match biological motifs and improve protein design in over 70% of cases.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the sparse latent representations learned jointly across layers faithfully approximate the model's full computational circuitry and that the identified circuits correspond to genuine structural and functional motifs rather than artifacts of the transcoder training.","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"1d102d2ef316c4efcc46f6ef42e8b6f7c786406baa1399ee6738ebcf5d26da8a"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.12026","created_at":"2026-05-18T03:09:23.608813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.12026v2","created_at":"2026-05-18T03:09:23.608813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.12026","created_at":"2026-05-18T03:09:23.608813+00:00"},{"alias_kind":"pith_short_12","alias_value":"65PALSYVPD7Q","created_at":"2026-05-18T12:33:37.589309+00:00"},{"alias_kind":"pith_short_16","alias_value":"65PALSYVPD7Q6GMT","created_at":"2026-05-18T12:33:37.589309+00:00"},{"alias_kind":"pith_short_8","alias_value":"65PALSYV","created_at":"2026-05-18T12:33:37.589309+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.16331","citing_title":"Retrieval and competition: how a protein foundation model starts a protein","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY","json":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY.json","graph_json":"https://pith.science/api/pith-number/65PALSYVPD7Q6GMTXKZI7PB2OY/graph.json","events_json":"https://pith.science/api/pith-number/65PALSYVPD7Q6GMTXKZI7PB2OY/events.json","paper":"https://pith.science/paper/65PALSYV"},"agent_actions":{"view_html":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY","download_json":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY.json","view_paper":"https://pith.science/paper/65PALSYV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.12026&json=true","fetch_graph":"https://pith.science/api/pith-number/65PALSYVPD7Q6GMTXKZI7PB2OY/graph.json","fetch_events":"https://pith.science/api/pith-number/65PALSYVPD7Q6GMTXKZI7PB2OY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY/action/storage_attestation","attest_author":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY/action/author_attestation","sign_citation":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY/action/citation_signature","submit_replication":"https://pith.science/pith/65PALSYVPD7Q6GMTXKZI7PB2OY/action/replication_record"}},"created_at":"2026-05-18T03:09:23.608813+00:00","updated_at":"2026-05-18T03:09:23.608813+00:00"}