{"work":{"id":"4f585692-8ef8-4f20-8f75-5e0e32647d76","openalex_id":null,"doi":null,"arxiv_id":null,"raw_key":"raw:989de7cc3808f81970d0426f","title":"Attention is all you need,","authors":null,"authors_text":"A","year":2017,"venue":null,"abstract":null,"external_url":null,"cited_by_count":null,"metadata_source":"raw_reference","metadata_fetched_at":"2026-07-10T14:07:06.849417+00:00","pith_arxiv_id":null,"created_at":"2026-05-12T13:26:36.102229+00:00","updated_at":"2026-07-10T14:07:06.849417+00:00","title_quality_ok":false,"display_title":"Attention is all you need","render_title":"Attention is all you need"},"hub":{"state":{"work_id":"4f585692-8ef8-4f20-8f75-5e0e32647d76","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":101,"external_cited_by_count":null,"distinct_field_count":18,"first_pith_cited_at":"2024-05-08T14:49:27+00:00","last_pith_cited_at":"2026-07-09T12:49:24+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T02:49:22.996028+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":8},{"context_role":"method","n":1}],"polarity_counts":[{"context_polarity":"background","n":8},{"context_polarity":"use_method","n":1}],"runs":{"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-26T07:16:39.545294+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":9},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":9},{"title":"Swin transformer: Hierarchical vision transformer using shifted windows","work_id":"7b33e4d0-6c5b-46e7-a3a4-589f80c21a6c","shared_citers":7},{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":5},{"title":"Deep residual learning for image recognition","work_id":"e36aa800-b42c-4273-94d4-d1cbbbbbef81","shared_citers":5},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":5},{"title":"Bert: Pre-training of deep bidirectional transformers for language understanding","work_id":"1df72c17-5cf2-4750-8a5f-749934f7ce53","shared_citers":4},{"title":"Denoising diffusion probabilistic models","work_id":"556bc6b5-6607-4a7d-871d-59b4cb93e3c9","shared_citers":4},{"title":"Fedformer: Frequency enhanced decomposed transformer for long-term series fore- casting","work_id":"503ed191-efd0-4a0e-a23b-4950f7adb7d9","shared_citers":4},{"title":"Informer: Beyond efficient transformer for long sequence time-series forecasting","work_id":"9e57c956-a900-4894-9ec1-1c4546e37b5e","shared_citers":4},{"title":"A Time Series is Worth 64 Words: Long-term Forecasting with Transformers","work_id":"d6d0a3ac-d695-4de0-ba2d-4e1d31ac8359","shared_citers":3},{"title":"Autoformer: Decomposition transformers with auto-correlation for long-term series forecasting","work_id":"5375df2c-806e-4da1-8f53-9f8bbe5e69ed","shared_citers":3},{"title":"Depth map prediction from a single image using a multi-scale deep network","work_id":"acced8ae-df4b-49ec-8e6d-a6fb76ee4997","shared_citers":3},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":3},{"title":"Efficiently Modeling Long Sequences with Structured State Spaces","work_id":"4150b761-b8bf-4d9b-a2f8-cb2d1b73d378","shared_citers":3},{"title":"Flashattention: Fast and memory-efficient exact attention with io-awareness,","work_id":"06f6ee67-14fa-4adc-855e-c598825d88e4","shared_citers":3},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":3},{"title":"iTransformer: Inverted Transformers Are Effective for Time Series Forecasting","work_id":"c1ea0b21-a74c-4315-a48c-4ef730cec588","shared_citers":3},{"title":"Layer Normalization","work_id":"20a2d720-0046-4c7c-bcd6-327ec8143f69","shared_citers":3},{"title":"Learning transferable visual models from natural language supervision","work_id":"0c972fd9-514c-4534-a4d4-07a79bb836a6","shared_citers":3},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":3},{"title":"Mamba: Linear-time sequence modeling with selective state spaces,","work_id":"bb24df99-6053-4040-9b46-b8b10cf38d19","shared_citers":3},{"title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","work_id":"4ee75248-1199-492c-a52f-6661e0f4adff","shared_citers":3},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":3}],"time_series":[{"n":1,"year":2024},{"n":11,"year":2025},{"n":60,"year":2026}],"dependency_candidates":[{"n":1,"role":"method","polarity":"use_method","paper_title":"TSNN: A Non-parametric and Interpretable Framework for Traffic Time Series Forecasting","primary_cat":"cs.LG","context_text":"of fitting complex functions even if the functions are implicit. Due to the natural capability of process sequence data, the neural networks based on Recurrent Neural Networks (RNNs) are widely used in time series forecasting [25]-[29]. How- ever, RNN-based models struggle with the iterative prediction process which accumulates errors. Later studies introduced transformer [30] and attention mechanisms to the time series models [31]. Zhou et al. proposed Informer [3] that adapts transformer for long-term time series forecasting by reducing the complexity of the self-attention mechanism, highlighting the main attention information. Informer successfully verifies the effectiveness of transformers in time series forecasting.","citing_arxiv_id":"2605.09208"}]},"error":null,"updated_at":"2026-05-26T07:16:51.278816+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"items":[{"title":"Qwen3 Technical Report","outcome":"unchanged","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"counts":{"fixed":0,"merged":0,"unchanged":1,"quarantined":0,"needs_external_resolution":0},"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-26T07:16:51.181852+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Attention is all you need","claims":[{"claim_text":"characteristics inherent in power load time series. Data-driven approaches based on artificial intelligence have become mainstream in recent years. Early methods centered on recurrent neural networks (RNNs) and convolutional neural networks (CNNs), which are adept at capturing temporal de- pendencies and inter-variable relationships [5]. With the advent of the Transformer architecture [6], attention-based models have advanced rapidly for time series forecasting, giving rise to numerous variants ","claim_type":"background","confidence":0.95,"evidence_strength":"citation_context"},{"claim_text":"However, these models still face challenges: their ability to explicitly model local interactions remains limited, and their interpretability is relatively weak. These drawbacks motivate our approach, which leverages physically grounded quantum walk dynamics to provide both richer local structural model- ing and improved interpretability. Formally, the self-attention mechanism in the Transformer framework [20] is defined as Attention (Q, K, V) = softmax (QKT √ d ) V(2) WhereQ, K, V∈R n×dare the ","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"Generating accurate, human-like motion requires ac- counting for variability in emotion and semantic emphasis, two aspects that remain underexplored. Computational efficiency is an additional requirement for real-time robotics applications. Model architectures have evolved from recurrent networks such as long short-term memory (LSTM) [16] to attention-based transformers [17]. Adversarial and diffusion-based methods have also been proposed to improve motion realism and diver- sity [2], [14], [18]","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"Neural Machine Translation (NMT) has emerged as a pow- erful end-to-end approach for automated translation, employ- ing a single neural network to directly model the probability of a target sentence given a source sentence [1]. In recent years, NMT models have significantly improved translation quality, accompanied by a substantial expansion in model scale. Since Transformer introduced [2], the parameter count of NMT models has grown exponentially. For instance, M2M-100 (12 billion parameters) [","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"after GEMM completion while the output tiles still reside in on- chip memory (L1/L2 caches or registers), we avoid costly global memory traffic. However, conventional normalization layers operate along the feature dimension, which often misaligns with the physical data layout of GEMM outputs. To address this, we proposesBlockNorm, a normalization approach inspired by GroupNorm [71] which is originally designed to apply normalization within individual channels of a feature map. In our version of ","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"bias parameters(γ, β)from conditional inputs, then modulates intermediate features viaγ⊙x+βto achieve lightweight conditional feature selection [7]. We observe that this channel- level modulation effectively adjusts feature weights with low overhead and good trainability. In contrast, attention-based cross-modal fusion typically relies on spatial weights or token- level interactions [9], [15]-[17], which increase computa- tional/parameter overhead and may complicate optimization in reinforcement","claim_type":"background","confidence":0.8,"evidence_strength":"citation_context"}],"why_cited":"Pith tracks Attention is all you need because it crossed a citation-hub threshold. Current citing contexts most often use it as background evidence (8 contexts).","role_counts":[{"n":8,"context_role":"background"},{"n":1,"context_role":"method"}]},"error":null,"updated_at":"2026-05-26T07:16:51.285481+00:00"}},"summary":{"title":"Attention is all you need","claims":[{"claim_text":"characteristics inherent in power load time series. Data-driven approaches based on artificial intelligence have become mainstream in recent years. Early methods centered on recurrent neural networks (RNNs) and convolutional neural networks (CNNs), which are adept at capturing temporal de- pendencies and inter-variable relationships [5]. With the advent of the Transformer architecture [6], attention-based models have advanced rapidly for time series forecasting, giving rise to numerous variants ","claim_type":"background","confidence":0.95,"evidence_strength":"citation_context"},{"claim_text":"However, these models still face challenges: their ability to explicitly model local interactions remains limited, and their interpretability is relatively weak. These drawbacks motivate our approach, which leverages physically grounded quantum walk dynamics to provide both richer local structural model- ing and improved interpretability. Formally, the self-attention mechanism in the Transformer framework [20] is defined as Attention (Q, K, V) = softmax (QKT √ d ) V(2) WhereQ, K, V∈R n×dare the ","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"Generating accurate, human-like motion requires ac- counting for variability in emotion and semantic emphasis, two aspects that remain underexplored. Computational efficiency is an additional requirement for real-time robotics applications. Model architectures have evolved from recurrent networks such as long short-term memory (LSTM) [16] to attention-based transformers [17]. Adversarial and diffusion-based methods have also been proposed to improve motion realism and diver- sity [2], [14], [18]","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"Neural Machine Translation (NMT) has emerged as a pow- erful end-to-end approach for automated translation, employ- ing a single neural network to directly model the probability of a target sentence given a source sentence [1]. In recent years, NMT models have significantly improved translation quality, accompanied by a substantial expansion in model scale. Since Transformer introduced [2], the parameter count of NMT models has grown exponentially. For instance, M2M-100 (12 billion parameters) [","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"after GEMM completion while the output tiles still reside in on- chip memory (L1/L2 caches or registers), we avoid costly global memory traffic. However, conventional normalization layers operate along the feature dimension, which often misaligns with the physical data layout of GEMM outputs. To address this, we proposesBlockNorm, a normalization approach inspired by GroupNorm [71] which is originally designed to apply normalization within individual channels of a feature map. In our version of ","claim_type":"background","confidence":0.9,"evidence_strength":"citation_context"},{"claim_text":"bias parameters(γ, β)from conditional inputs, then modulates intermediate features viaγ⊙x+βto achieve lightweight conditional feature selection [7]. We observe that this channel- level modulation effectively adjusts feature weights with low overhead and good trainability. In contrast, attention-based cross-modal fusion typically relies on spatial weights or token- level interactions [9], [15]-[17], which increase computa- tional/parameter overhead and may complicate optimization in reinforcement","claim_type":"background","confidence":0.8,"evidence_strength":"citation_context"}],"why_cited":"Pith tracks Attention is all you need because it crossed a citation-hub threshold. Current citing contexts most often use it as background evidence (8 contexts).","role_counts":[{"n":8,"context_role":"background"},{"n":1,"context_role":"method"}]},"graph":{"co_cited":[{"title":"Adam: A Method for Stochastic Optimization","work_id":"1910796d-9b52-4683-bf5c-de9632c1028b","shared_citers":9},{"title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","work_id":"e96730e3-129b-4db6-b981-15ab7932e297","shared_citers":9},{"title":"Swin transformer: Hierarchical vision transformer using shifted windows","work_id":"7b33e4d0-6c5b-46e7-a3a4-589f80c21a6c","shared_citers":7},{"title":"Decoupled Weight Decay Regularization","work_id":"07ef7360-d385-4033-83f7-8384a6325204","shared_citers":5},{"title":"Deep residual learning for image recognition","work_id":"e36aa800-b42c-4273-94d4-d1cbbbbbef81","shared_citers":5},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":5},{"title":"Bert: Pre-training of deep bidirectional transformers for language understanding","work_id":"1df72c17-5cf2-4750-8a5f-749934f7ce53","shared_citers":4},{"title":"Denoising diffusion probabilistic models","work_id":"556bc6b5-6607-4a7d-871d-59b4cb93e3c9","shared_citers":4},{"title":"Fedformer: Frequency enhanced decomposed transformer for long-term series fore- casting","work_id":"503ed191-efd0-4a0e-a23b-4950f7adb7d9","shared_citers":4},{"title":"Informer: Beyond efficient transformer for long sequence time-series forecasting","work_id":"9e57c956-a900-4894-9ec1-1c4546e37b5e","shared_citers":4},{"title":"A Time Series is Worth 64 Words: Long-term Forecasting with Transformers","work_id":"d6d0a3ac-d695-4de0-ba2d-4e1d31ac8359","shared_citers":3},{"title":"Autoformer: Decomposition transformers with auto-correlation for long-term series forecasting","work_id":"5375df2c-806e-4da1-8f53-9f8bbe5e69ed","shared_citers":3},{"title":"Depth map prediction from a single image using a multi-scale deep network","work_id":"acced8ae-df4b-49ec-8e6d-a6fb76ee4997","shared_citers":3},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":3},{"title":"Efficiently Modeling Long Sequences with Structured State Spaces","work_id":"4150b761-b8bf-4d9b-a2f8-cb2d1b73d378","shared_citers":3},{"title":"Flashattention: Fast and memory-efficient exact attention with io-awareness,","work_id":"06f6ee67-14fa-4adc-855e-c598825d88e4","shared_citers":3},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":3},{"title":"iTransformer: Inverted Transformers Are Effective for Time Series Forecasting","work_id":"c1ea0b21-a74c-4315-a48c-4ef730cec588","shared_citers":3},{"title":"Layer Normalization","work_id":"20a2d720-0046-4c7c-bcd6-327ec8143f69","shared_citers":3},{"title":"Learning transferable visual models from natural language supervision","work_id":"0c972fd9-514c-4534-a4d4-07a79bb836a6","shared_citers":3},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":3},{"title":"Mamba: Linear-time sequence modeling with selective state spaces,","work_id":"bb24df99-6053-4040-9b46-b8b10cf38d19","shared_citers":3},{"title":"Mamba: Linear-Time Sequence Modeling with Selective State Spaces","work_id":"4ee75248-1199-492c-a52f-6661e0f4adff","shared_citers":3},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":3}],"time_series":[{"n":1,"year":2024},{"n":11,"year":2025},{"n":60,"year":2026}],"dependency_candidates":[{"n":1,"role":"method","polarity":"use_method","paper_title":"TSNN: A Non-parametric and Interpretable Framework for Traffic Time Series Forecasting","primary_cat":"cs.LG","context_text":"of fitting complex functions even if the functions are implicit. Due to the natural capability of process sequence data, the neural networks based on Recurrent Neural Networks (RNNs) are widely used in time series forecasting [25]-[29]. How- ever, RNN-based models struggle with the iterative prediction process which accumulates errors. Later studies introduced transformer [30] and attention mechanisms to the time series models [31]. Zhou et al. proposed Informer [3] that adapts transformer for long-term time series forecasting by reducing the complexity of the self-attention mechanism, highlighting the main attention information. Informer successfully verifies the effectiveness of transformers in time series forecasting.","citing_arxiv_id":"2605.09208"}]},"authors":[]}}