{"as_of":"2026-08-09T10:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d85ac372bd03205909a9e5c34bad04e8116e650a51e3313fc67442bc8e2bdb53","coverage":[{"denominator":12,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T20:25:25.335296Z","state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.05076/citation-record","integrity":"/paper/2502.05076/integrity","json":"/paper/2502.05076/citation-record.json","paper":"/paper/2502.05076"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.14316","last_updated":"2024-07-16T10:22:51Z","snapshot_observed_at":"2026-08-09T00:44:37.308469Z","submitted_at":"2023-09-25T17:37:20Z","title":"Physics of Language Models: Part 3.1, Knowledge Storage and Extraction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14316","snapshot_observed_at":"2026-08-08T20:25:25.280155Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.280155Z"},"links":{"cited_paper":"/paper/2309.14316","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:74400c0bbca97c923fc7b40823206f619fe432113713c3bad4078e3a48442ad7","observation_id":"60c5cb77-9cd5-42ae-a972-8d37de90996d","resolution":{"observed_at":"2026-08-08T20:25:25.280155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05405","last_updated":"2024-04-08T11:11:31Z","snapshot_observed_at":"2026-08-05T03:30:28.158531Z","submitted_at":"2024-04-08T11:11:31Z","title":"Physics of Language Models: Part 3.3, Knowledge Capacity Scaling Laws","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05405","snapshot_observed_at":"2026-08-08T20:25:25.286028Z","title":"Physics of language models: Part 3.3, knowledge capacity scaling laws","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.286028Z"},"links":{"cited_paper":"/paper/2404.05405","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:b9d824ab22d356a1a08821f504936ec579df6cf7c13ce4511b443fae356e0de0","observation_id":"8807e8bb-2dfb-425a-890a-c20c02444a30","resolution":{"observed_at":"2026-08-08T20:25:25.286028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07321","last_updated":"2024-02-11T22:58:49Z","snapshot_observed_at":"2026-08-08T20:05:22.419832Z","submitted_at":"2024-02-11T22:58:49Z","title":"Summing Up the Facts: Additive Mechanisms Behind Factual Recall in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07321","snapshot_observed_at":"2026-08-08T20:25:25.291312Z","title":"Summing up the facts: Additive mechanisms behind factual recall in LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.291312Z"},"links":{"cited_paper":"/paper/2402.07321","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:f2750e95e19a574d269e104930a7e95ef1ef3913156a66add87bdbbf260ea95a","observation_id":"1299dd8a-6a16-473b-90d7-9f8048ac030b","resolution":{"observed_at":"2026-08-08T20:25:25.291312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:25:25.296297Z","title":"A mathematical framework for transformer circuits","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.296297Z"},"links":{"citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:a626464177a8c8208d9a38f61b2e6e5610db67b6ab008d4d9f1efea065cfe00a","observation_id":"0f9c35b8-c566-4881-a68b-9054ba5c7e74","resolution":{"observed_at":"2026-08-08T20:25:25.296297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2012.14913","last_updated":"2021-09-05T17:32:27Z","snapshot_observed_at":"2026-08-07T23:50:29.977330Z","submitted_at":"2020-12-29T19:12:05Z","title":"Transformer Feed-Forward Layers Are Key-Value Memories","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2012.14913","snapshot_observed_at":"2026-08-08T20:25:25.301583Z","title":"Transformer feed-forward layers are key-value memories.arXiv preprint arXiv:2012.14913, 2020","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.301583Z"},"links":{"cited_paper":"/paper/2012.14913","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:d33317860e9e51d4e82c20581989a2db0a9217af1ace00b4e7a11f8cd5be0124","observation_id":"a3ef1b7a-3bd0-422e-8172-e43103011287","resolution":{"observed_at":"2026-08-08T20:25:25.301583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:25:25.523894Z","title":"Tensor rank is NP-complete.Journal of algorithms, 11(4):644– 654, 1990","venue":null,"work_id":"43384896-bce0-431e-bbb4-3dcc5e79eb05","year":1990},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.306914Z"},"links":{"citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:9a0a927969497c93652f7f342b4af8147bb78de5ac039af44f6b01d728f94747","observation_id":"477fe997-f862-4438-9534-c62270dfa062","resolution":{"observed_at":"2026-08-08T20:25:25.528832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15720","last_updated":"2024-06-22T03:32:09Z","snapshot_observed_at":"2026-07-06T18:35:17.953523Z","submitted_at":"2024-06-22T03:32:09Z","title":"Scaling Laws for Fact Memorization of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15720","snapshot_observed_at":"2026-08-08T20:25:25.312186Z","title":"Scaling laws for fact memorization of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.312186Z"},"links":{"cited_paper":"/paper/2406.15720","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:b516a5153c4186337965c579dd861c6d7200a681441133aaf5acb4e393ac917e","observation_id":"a138a281-cd56-4c8c-9641-a145c6b01daa","resolution":{"observed_at":"2026-08-08T20:25:25.312186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06538","last_updated":"2024-12-09T14:48:14Z","snapshot_observed_at":"2026-08-05T08:13:09.313977Z","submitted_at":"2024-12-09T14:48:14Z","title":"Understanding Factual Recall in Transformers via Associative Memories","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06538","snapshot_observed_at":"2026-08-08T20:25:25.316854Z","title":"Understanding fac- tual recall in transformers via associative memories","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.316854Z"},"links":{"cited_paper":"/paper/2412.06538","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:b8a2b7f7d8556aeab5b3c02c86e6f18958d48bf8cbe5eec8c81c5490834f6a28","observation_id":"ada139e0-0b38-48bd-8eff-334963a6f839","resolution":{"observed_at":"2026-08-08T20:25:25.316854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:25:25.507224Z","title":"Tensor rank is hard to approximate","venue":null,"work_id":"7647f993-0d80-42cd-bc53-f18afdd0701f","year":2018},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.321845Z"},"links":{"citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:511fbcde4fb9190fa3eb211e79402866e643ed43c7651ca35224348a24cea66d","observation_id":"9afedfad-32c4-4bb0-a4c7-289d22a7a2e5","resolution":{"observed_at":"2026-08-08T20:25:25.512580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T20:25:25.490370Z","title":"Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N","venue":null,"work_id":"bad61ce8-f514-4b30-9a16-9ccd7dc71a62","year":2017},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.326351Z"},"links":{"citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:e1b95e4c0bc0f86aab9197dfa65b32551d9b13c83a752257554f1214cb6e8b92","observation_id":"2f06bb40-e7ff-4fbf-8c07-582594f841d0","resolution":{"observed_at":"2026-08-08T20:25:25.496692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.03953","last_updated":"2018-03-02T20:20:52Z","snapshot_observed_at":"2026-08-04T06:22:23.897655Z","submitted_at":"2017-11-10T18:29:00Z","title":"Breaking the Softmax Bottleneck: A High-Rank RNN Language Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.03953","snapshot_observed_at":"2026-08-08T20:25:25.330692Z","title":"Breaking the softmax bottleneck: A high-rank RNN language model","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.330692Z"},"links":{"cited_paper":"/paper/1711.03953","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:c6f587911019f2890baa20b7c74b241c4ff3ef855ada969580cd5752e581bef6","observation_id":"8f478b2b-1301-4da4-af95-bb85149dbf9f","resolution":{"observed_at":"2026-08-08T20:25:25.330692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17969","last_updated":"2025-01-03T16:41:37Z","snapshot_observed_at":"2026-08-08T01:27:33.186418Z","submitted_at":"2024-05-28T08:56:33Z","title":"Knowledge Circuits in Pretrained Transformers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17969","snapshot_observed_at":"2026-08-08T20:25:25.335296Z","title":"Knowledge circuits in pretrained transformers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T20:25:25.335296Z"},"links":{"cited_paper":"/paper/2405.17969","citing_paper":"/paper/2502.05076"},"observation_digest":"sha256:2474353da0bc96de3accbceaaa15bee503f80e582408796ea5fdc71498486ba3","observation_id":"fd82dd60-88e8-4f6e-9787-82e3c8b5dd33","resolution":{"observed_at":"2026-08-08T20:25:25.335296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.05076","last_updated":"2025-02-07T16:50:27Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-09T00:45:06.230019Z","submitted_at":"2025-02-07T16:50:27Z","title":"Paying Attention to Facts: Quantifying the Knowledge Capacity of Attention Layers"},"reference_resolution":{"displayed":12,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":9,"verified_exact":0,"verified_fuzzy":3},"total_outbound_references":12},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 12 of 12 outbound references and 0 inbound Pith citation observations for arXiv:2502.05076."}