{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:RFNCO23RL4UEZIIGXGFLHUL4FL","short_pith_number":"pith:RFNCO23R","schema_version":"1.0","canonical_sha256":"895a276b715f284ca106b98ab3d17c2ad6fcbd2e8d220507df5eb0c2302e5d6f","source":{"kind":"arxiv","id":"2006.04710","version":2},"attestation_state":"computed","paper":{"title":"The Lipschitz Constant of Self-Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Andriy Mnih, George Papamakarios, Hyunjik Kim","submitted_at":"2020-06-08T16:08:38Z","abstract_excerpt":"Lipschitz constants of neural networks have been explored in various contexts in deep learning, such as provable adversarial robustness, estimating Wasserstein distance, stabilising training of GANs, and formulating invertible neural networks. Such works have focused on bounding the Lipschitz constant of fully connected or convolutional networks, composed of linear maps and pointwise non-linearities. In this paper, we investigate the Lipschitz constant of self-attention, a non-linear neural network module widely used in sequence modelling. We prove that the standard dot-product self-attention "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.04710","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2020-06-08T16:08:38Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6bda48ee9f77fb9ec2797787fc1d6307c1e701210df804576d19ff66c67af50e","abstract_canon_sha256":"0d407eaaf362c5dea8a3f83e628612c4da9e5f59d89bfc1fd4a5a709120ee160"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:47:39.093061Z","signature_b64":"xjIGDrncFi399Tx4Wt8wkOAxkw2NxJrrkFiJRRhcT/jfIl44zxn6AHtgD5wi+nUpOvLhef3+rR7/pakIXVMdDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"895a276b715f284ca106b98ab3d17c2ad6fcbd2e8d220507df5eb0c2302e5d6f","last_reissued_at":"2026-07-05T02:47:39.092646Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:47:39.092646Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Lipschitz Constant of Self-Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Andriy Mnih, George Papamakarios, Hyunjik Kim","submitted_at":"2020-06-08T16:08:38Z","abstract_excerpt":"Lipschitz constants of neural networks have been explored in various contexts in deep learning, such as provable adversarial robustness, estimating Wasserstein distance, stabilising training of GANs, and formulating invertible neural networks. Such works have focused on bounding the Lipschitz constant of fully connected or convolutional networks, composed of linear maps and pointwise non-linearities. In this paper, we investigate the Lipschitz constant of self-attention, a non-linear neural network module widely used in sequence modelling. We prove that the standard dot-product self-attention "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.04710","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.04710/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.04710","created_at":"2026-07-05T02:47:39.092710+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.04710v2","created_at":"2026-07-05T02:47:39.092710+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.04710","created_at":"2026-07-05T02:47:39.092710+00:00"},{"alias_kind":"pith_short_12","alias_value":"RFNCO23RL4UE","created_at":"2026-07-05T02:47:39.092710+00:00"},{"alias_kind":"pith_short_16","alias_value":"RFNCO23RL4UEZIIG","created_at":"2026-07-05T02:47:39.092710+00:00"},{"alias_kind":"pith_short_8","alias_value":"RFNCO23R","created_at":"2026-07-05T02:47:39.092710+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.07925","citing_title":"Sinkhorn doubly stochastic attention rank decay analysis","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL","json":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL.json","graph_json":"https://pith.science/api/pith-number/RFNCO23RL4UEZIIGXGFLHUL4FL/graph.json","events_json":"https://pith.science/api/pith-number/RFNCO23RL4UEZIIGXGFLHUL4FL/events.json","paper":"https://pith.science/paper/RFNCO23R"},"agent_actions":{"view_html":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL","download_json":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL.json","view_paper":"https://pith.science/paper/RFNCO23R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.04710&json=true","fetch_graph":"https://pith.science/api/pith-number/RFNCO23RL4UEZIIGXGFLHUL4FL/graph.json","fetch_events":"https://pith.science/api/pith-number/RFNCO23RL4UEZIIGXGFLHUL4FL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL/action/storage_attestation","attest_author":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL/action/author_attestation","sign_citation":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL/action/citation_signature","submit_replication":"https://pith.science/pith/RFNCO23RL4UEZIIGXGFLHUL4FL/action/replication_record"}},"created_at":"2026-07-05T02:47:39.092710+00:00","updated_at":"2026-07-05T02:47:39.092710+00:00"}