{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UDMHRLJMMQ7G23KAV3RLTMAM3H","short_pith_number":"pith:UDMHRLJM","schema_version":"1.0","canonical_sha256":"a0d878ad2c643e6d6d40aee2b9b00cd9d5e0c15c7e109111098b5cea46f7d08b","source":{"kind":"arxiv","id":"2501.12900","version":3},"attestation_state":"computed","paper":{"title":"Unified CNNs and transformers underlying learning mechanism reveals multi-head attention modus vivendi","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Ella Koresh, Ido Kanter, Ronit D. Gross, Tal Halevi, Yarden Tzach, Yuval Meir","submitted_at":"2025-01-22T14:19:48Z","abstract_excerpt":"Convolutional neural networks (CNNs) evaluate short-range correlations in input images which progress along the layers, whereas vision transformer (ViT) architectures evaluate long-range correlations, using repeated transformer encoders composed of fully connected layers. Both are designed to solve complex classification tasks but from different perspectives. This study demonstrates that CNNs and ViT architectures stem from a unified underlying learning mechanism, which quantitatively measures the single-nodal performance (SNP) of each node in feedforward (FF) and multi-head attention (MHA) su"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.12900","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-22T14:19:48Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"22b41c12f190061e5a7f339c5d5552626c408c0c4758dac6274e92f427b3f2a5","abstract_canon_sha256":"883b484c664a5478cc3f6c9e3dfe1d5997d5d3743603a2ece9add5444915aa7b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:40.740074Z","signature_b64":"pQlK1+tLhR22o8K1Q8QC5rUTaez+EacSgcVTrm9pfP0S+raZ2zOMCWHvhITyd5jgmZotoFEtTL8qZC7c3oiiCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0d878ad2c643e6d6d40aee2b9b00cd9d5e0c15c7e109111098b5cea46f7d08b","last_reissued_at":"2026-07-05T10:46:40.739616Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:40.739616Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unified CNNs and transformers underlying learning mechanism reveals multi-head attention modus vivendi","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Ella Koresh, Ido Kanter, Ronit D. Gross, Tal Halevi, Yarden Tzach, Yuval Meir","submitted_at":"2025-01-22T14:19:48Z","abstract_excerpt":"Convolutional neural networks (CNNs) evaluate short-range correlations in input images which progress along the layers, whereas vision transformer (ViT) architectures evaluate long-range correlations, using repeated transformer encoders composed of fully connected layers. Both are designed to solve complex classification tasks but from different perspectives. This study demonstrates that CNNs and ViT architectures stem from a unified underlying learning mechanism, which quantitatively measures the single-nodal performance (SNP) of each node in feedforward (FF) and multi-head attention (MHA) su"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.12900","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.12900/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.12900","created_at":"2026-07-05T10:46:40.739665+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.12900v3","created_at":"2026-07-05T10:46:40.739665+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.12900","created_at":"2026-07-05T10:46:40.739665+00:00"},{"alias_kind":"pith_short_12","alias_value":"UDMHRLJMMQ7G","created_at":"2026-07-05T10:46:40.739665+00:00"},{"alias_kind":"pith_short_16","alias_value":"UDMHRLJMMQ7G23KA","created_at":"2026-07-05T10:46:40.739665+00:00"},{"alias_kind":"pith_short_8","alias_value":"UDMHRLJM","created_at":"2026-07-05T10:46:40.739665+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.03407","citing_title":"Learning Mechanism Underlying NLP Pre-Training and Fine-Tuning","ref_index":18,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H","json":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H.json","graph_json":"https://pith.science/api/pith-number/UDMHRLJMMQ7G23KAV3RLTMAM3H/graph.json","events_json":"https://pith.science/api/pith-number/UDMHRLJMMQ7G23KAV3RLTMAM3H/events.json","paper":"https://pith.science/paper/UDMHRLJM"},"agent_actions":{"view_html":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H","download_json":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H.json","view_paper":"https://pith.science/paper/UDMHRLJM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.12900&json=true","fetch_graph":"https://pith.science/api/pith-number/UDMHRLJMMQ7G23KAV3RLTMAM3H/graph.json","fetch_events":"https://pith.science/api/pith-number/UDMHRLJMMQ7G23KAV3RLTMAM3H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H/action/storage_attestation","attest_author":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H/action/author_attestation","sign_citation":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H/action/citation_signature","submit_replication":"https://pith.science/pith/UDMHRLJMMQ7G23KAV3RLTMAM3H/action/replication_record"}},"created_at":"2026-07-05T10:46:40.739665+00:00","updated_at":"2026-07-05T10:46:40.739665+00:00"}