{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZUSDYLWTFIDNTAQ3OU2BNWNPHC","short_pith_number":"pith:ZUSDYLWT","schema_version":"1.0","canonical_sha256":"cd243c2ed32a06d9821b753416d9af38b07ff2c063ced3a7a1c9677b3a714794","source":{"kind":"arxiv","id":"2310.19380","version":4},"attestation_state":"computed","paper":{"title":"TransXNet: Learning Both Global and Local Dynamics with a Dual Dynamic Token Mixer for Visual Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuan Wu, Hong-Yu Zhou, Meng Lou, Shu Zhang, Sibei Yang, Yizhou Yu","submitted_at":"2023-10-30T09:35:56Z","abstract_excerpt":"Recent studies have integrated convolutions into transformers to introduce inductive bias and improve generalization performance. However, the static nature of conventional convolution prevents it from dynamically adapting to input variations, resulting in a representation discrepancy between convolution and self-attention as the latter computes attention maps dynamically. Furthermore, when stacking token mixers that consist of convolution and self-attention to form a deep network, the static nature of convolution hinders the fusion of features previously generated by self-attention into convo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.19380","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-10-30T09:35:56Z","cross_cats_sorted":[],"title_canon_sha256":"273ae3823880f17ae44047f13f45b0138d1844801de2b8d89bf68bb329c2ed25","abstract_canon_sha256":"8c1189f8736cf461a114d7c6e7eaa60acd2174912666e4f41154b74dbcff0b89"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:43.350121Z","signature_b64":"Oz5f50C2vHfwcljzl4IrmnzqWMKzUs1GiHh/EUO8yggNpECoPkDJQEJpZGjGj40vxA3Ax5DVXCNhENTH4OxzAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd243c2ed32a06d9821b753416d9af38b07ff2c063ced3a7a1c9677b3a714794","last_reissued_at":"2026-07-05T10:53:43.349597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:43.349597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TransXNet: Learning Both Global and Local Dynamics with a Dual Dynamic Token Mixer for Visual Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuan Wu, Hong-Yu Zhou, Meng Lou, Shu Zhang, Sibei Yang, Yizhou Yu","submitted_at":"2023-10-30T09:35:56Z","abstract_excerpt":"Recent studies have integrated convolutions into transformers to introduce inductive bias and improve generalization performance. However, the static nature of conventional convolution prevents it from dynamically adapting to input variations, resulting in a representation discrepancy between convolution and self-attention as the latter computes attention maps dynamically. Furthermore, when stacking token mixers that consist of convolution and self-attention to form a deep network, the static nature of convolution hinders the fusion of features previously generated by self-attention into convo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.19380","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.19380/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.19380","created_at":"2026-07-05T10:53:43.349666+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.19380v4","created_at":"2026-07-05T10:53:43.349666+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.19380","created_at":"2026-07-05T10:53:43.349666+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZUSDYLWTFIDN","created_at":"2026-07-05T10:53:43.349666+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZUSDYLWTFIDNTAQ3","created_at":"2026-07-05T10:53:43.349666+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZUSDYLWT","created_at":"2026-07-05T10:53:43.349666+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.19676","citing_title":"Optimizing Local-Global Dependencies for Accurate 3D Human Pose Estimation","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC","json":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC.json","graph_json":"https://pith.science/api/pith-number/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/graph.json","events_json":"https://pith.science/api/pith-number/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/events.json","paper":"https://pith.science/paper/ZUSDYLWT"},"agent_actions":{"view_html":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC","download_json":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC.json","view_paper":"https://pith.science/paper/ZUSDYLWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.19380&json=true","fetch_graph":"https://pith.science/api/pith-number/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/graph.json","fetch_events":"https://pith.science/api/pith-number/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/action/storage_attestation","attest_author":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/action/author_attestation","sign_citation":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/action/citation_signature","submit_replication":"https://pith.science/pith/ZUSDYLWTFIDNTAQ3OU2BNWNPHC/action/replication_record"}},"created_at":"2026-07-05T10:53:43.349666+00:00","updated_at":"2026-07-05T10:53:43.349666+00:00"}