{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:76DWL4H7FWXRTBETH3DMEIMIVD","short_pith_number":"pith:76DWL4H7","schema_version":"1.0","canonical_sha256":"ff8765f0ff2daf1984933ec6c22188a8d70e85d88554f52d5dab5662927385de","source":{"kind":"arxiv","id":"2410.12101","version":2},"attestation_state":"computed","paper":{"title":"The Persian Rug: solving toy models of superposition using large-scale symmetries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.AI"],"primary_cat":"cs.LG","authors_text":"Aditya Cowsik, Alex Infanger, Kfir Dolev","submitted_at":"2024-10-15T22:52:45Z","abstract_excerpt":"We present a complete mechanistic description of the algorithm learned by a minimal non-linear sparse data autoencoder in the limit of large input dimension. The model, originally presented in arXiv:2209.10652, compresses sparse data vectors through a linear layer and decompresses using another linear layer followed by a ReLU activation. We notice that when the data is permutation symmetric (no input feature is privileged) large models reliably learn an algorithm that is sensitive to individual weights only through their large-scale statistics. For these models, the loss function becomes analy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12101","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-15T22:52:45Z","cross_cats_sorted":["cond-mat.dis-nn","cs.AI"],"title_canon_sha256":"4999b5d68da92cd61fe06922f793f023437d8a54547e853503df1b5681cb8fb9","abstract_canon_sha256":"7b2fe5b4a597827404afb855de22d12a5967f3d3ebe56c2fd8d1a281043a75d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:00.447736Z","signature_b64":"/wBE713c6XlgSednWkB0aP/JhG+p7+Di0IaIl4ipTMu0R9lTKl43773vzLVstlNZkb+l30jFLwje3cF7pTQACg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff8765f0ff2daf1984933ec6c22188a8d70e85d88554f52d5dab5662927385de","last_reissued_at":"2026-07-05T09:24:00.447200Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:00.447200Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Persian Rug: solving toy models of superposition using large-scale symmetries","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.AI"],"primary_cat":"cs.LG","authors_text":"Aditya Cowsik, Alex Infanger, Kfir Dolev","submitted_at":"2024-10-15T22:52:45Z","abstract_excerpt":"We present a complete mechanistic description of the algorithm learned by a minimal non-linear sparse data autoencoder in the limit of large input dimension. The model, originally presented in arXiv:2209.10652, compresses sparse data vectors through a linear layer and decompresses using another linear layer followed by a ReLU activation. We notice that when the data is permutation symmetric (no input feature is privileged) large models reliably learn an algorithm that is sensitive to individual weights only through their large-scale statistics. For these models, the loss function becomes analy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12101","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12101/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12101","created_at":"2026-07-05T09:24:00.447265+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12101v2","created_at":"2026-07-05T09:24:00.447265+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12101","created_at":"2026-07-05T09:24:00.447265+00:00"},{"alias_kind":"pith_short_12","alias_value":"76DWL4H7FWXR","created_at":"2026-07-05T09:24:00.447265+00:00"},{"alias_kind":"pith_short_16","alias_value":"76DWL4H7FWXRTBET","created_at":"2026-07-05T09:24:00.447265+00:00"},{"alias_kind":"pith_short_8","alias_value":"76DWL4H7","created_at":"2026-07-05T09:24:00.447265+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18538","citing_title":"Effects of sparsity and superposition on loss in simple autoencoders","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01227","citing_title":"The Lattice Representation Hypothesis of Large Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD","json":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD.json","graph_json":"https://pith.science/api/pith-number/76DWL4H7FWXRTBETH3DMEIMIVD/graph.json","events_json":"https://pith.science/api/pith-number/76DWL4H7FWXRTBETH3DMEIMIVD/events.json","paper":"https://pith.science/paper/76DWL4H7"},"agent_actions":{"view_html":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD","download_json":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD.json","view_paper":"https://pith.science/paper/76DWL4H7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12101&json=true","fetch_graph":"https://pith.science/api/pith-number/76DWL4H7FWXRTBETH3DMEIMIVD/graph.json","fetch_events":"https://pith.science/api/pith-number/76DWL4H7FWXRTBETH3DMEIMIVD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD/action/storage_attestation","attest_author":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD/action/author_attestation","sign_citation":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD/action/citation_signature","submit_replication":"https://pith.science/pith/76DWL4H7FWXRTBETH3DMEIMIVD/action/replication_record"}},"created_at":"2026-07-05T09:24:00.447265+00:00","updated_at":"2026-07-05T09:24:00.447265+00:00"}