{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PDIB3KYYHJEDXIUWFQKS4UWRCK","short_pith_number":"pith:PDIB3KYY","schema_version":"1.0","canonical_sha256":"78d01dab183a483ba2962c152e52d112bb700e0626e56e52977eca568a21ee07","source":{"kind":"arxiv","id":"2507.13338","version":1},"attestation_state":"computed","paper":{"title":"Training Transformers with Enforced Lipschitz Constants","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrii Zahorodnii, Franz Cesista, Jeremy Bernstein, Laker Newhouse, Phillip Isola, R. Preston Hess","submitted_at":"2025-07-17T17:55:00Z","abstract_excerpt":"Neural networks are often highly sensitive to input and weight perturbations. This sensitivity has been linked to pathologies such as vulnerability to adversarial examples, divergent training, and overfitting. To combat these problems, past research has looked at building neural networks entirely from Lipschitz components. However, these techniques have not matured to the point where researchers have trained a modern architecture such as a transformer with a Lipschitz certificate enforced beyond initialization. To explore this gap, we begin by developing and benchmarking novel, computationally"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.13338","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-17T17:55:00Z","cross_cats_sorted":[],"title_canon_sha256":"e990b6f436ca04b8b8b9053f00d527c3e9a4834ee71860c9d4ff4954d991258a","abstract_canon_sha256":"b6d286005ef51248a4ffd76e64f44fb600fb4efaeb6ae0810e183f7e28707a56"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:38:59.454974Z","signature_b64":"VXxhzk3DdChqm+KYVBW+6nR5G0aMYSZURYhtm/4Rc38kszGFoRJHVwpiItnJYc9NFCHla56cygMA1WPSp3IdBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78d01dab183a483ba2962c152e52d112bb700e0626e56e52977eca568a21ee07","last_reissued_at":"2026-07-05T11:38:59.454528Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:38:59.454528Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Transformers with Enforced Lipschitz Constants","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrii Zahorodnii, Franz Cesista, Jeremy Bernstein, Laker Newhouse, Phillip Isola, R. Preston Hess","submitted_at":"2025-07-17T17:55:00Z","abstract_excerpt":"Neural networks are often highly sensitive to input and weight perturbations. This sensitivity has been linked to pathologies such as vulnerability to adversarial examples, divergent training, and overfitting. To combat these problems, past research has looked at building neural networks entirely from Lipschitz components. However, these techniques have not matured to the point where researchers have trained a modern architecture such as a transformer with a Lipschitz certificate enforced beyond initialization. To explore this gap, we begin by developing and benchmarking novel, computationally"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.13338","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.13338/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.13338","created_at":"2026-07-05T11:38:59.454576+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.13338v1","created_at":"2026-07-05T11:38:59.454576+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.13338","created_at":"2026-07-05T11:38:59.454576+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDIB3KYYHJED","created_at":"2026-07-05T11:38:59.454576+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDIB3KYYHJEDXIUW","created_at":"2026-07-05T11:38:59.454576+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDIB3KYY","created_at":"2026-07-05T11:38:59.454576+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26929","citing_title":"When Muon Optimizer Meets Adversarial Training: A Theoretical and Empirical Study","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27733","citing_title":"Can Entry-Wise Clipping Give Spectral Control of Stochastic Gradients?","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.22002","citing_title":"Understanding Rate-Distortion Performance in Distributed Transformer Inference","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04155","citing_title":"The Geometric Alignment Tax: Tokenization vs. Continuous Geometry in Scientific Foundation Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11850","citing_title":"Constrained Stochastic Spectral Preconditioning Converges for Nonconvex Objectives","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08134","citing_title":"DARE: Diffusion Language Model Activation Reuse for Efficient Inference","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK","json":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK.json","graph_json":"https://pith.science/api/pith-number/PDIB3KYYHJEDXIUWFQKS4UWRCK/graph.json","events_json":"https://pith.science/api/pith-number/PDIB3KYYHJEDXIUWFQKS4UWRCK/events.json","paper":"https://pith.science/paper/PDIB3KYY"},"agent_actions":{"view_html":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK","download_json":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK.json","view_paper":"https://pith.science/paper/PDIB3KYY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.13338&json=true","fetch_graph":"https://pith.science/api/pith-number/PDIB3KYYHJEDXIUWFQKS4UWRCK/graph.json","fetch_events":"https://pith.science/api/pith-number/PDIB3KYYHJEDXIUWFQKS4UWRCK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK/action/storage_attestation","attest_author":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK/action/author_attestation","sign_citation":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK/action/citation_signature","submit_replication":"https://pith.science/pith/PDIB3KYYHJEDXIUWFQKS4UWRCK/action/replication_record"}},"created_at":"2026-07-05T11:38:59.454576+00:00","updated_at":"2026-07-05T11:38:59.454576+00:00"}