{"as_of":"2026-08-08T16:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e8d17768a57f8b7b75540d8fb2b73ea2da5a1d1c55d50f0a9473f881edb46348","coverage":[{"denominator":89,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":89,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:57:50.562642Z","state":"measured"},{"denominator":89,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":89,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.01414/citation-record","integrity":"/paper/2507.01414/integrity","json":"/paper/2507.01414/citation-record.json","paper":"/paper/2507.01414"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.12973","last_updated":"2024-01-30T18:59:34Z","snapshot_observed_at":"2026-07-06T17:19:31.847730Z","submitted_at":"2024-01-23T18:59:21Z","title":"In-Context Language Learning: Architectures and Algorithms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.12973","snapshot_observed_at":"2026-08-06T20:57:47.325633Z","title":"In-context language learning: Architectures and algorithms.arXiv preprint arXiv:2401.12973, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.325633Z"},"links":{"cited_paper":"/paper/2401.12973","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:305bdb546da51daf17683f5b61d970327984a939c2dfe3b625c433042ec35146","observation_id":"5d63445b-9904-4aa1-9fcd-352603307b18","resolution":{"observed_at":"2026-08-06T20:57:47.325633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.391018Z","title":"Lepori, Jack Merullo, and Ellie Pavlick","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.391018Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:5e31a7e06a3adeec9205bfa8f0a83bb2f5ec8d50f129f777eebee2ed71afa9da","observation_id":"68b19d44-e34d-4448-a1b7-0f8fc71f69b8","resolution":{"observed_at":"2026-08-06T20:57:47.391018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04927","last_updated":"2023-12-08T09:44:25Z","snapshot_observed_at":"2026-08-05T07:11:50.808005Z","submitted_at":"2023-12-08T09:44:25Z","title":"Zoology: Measuring and Improving Recall in Efficient Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04927","snapshot_observed_at":"2026-08-06T20:57:47.494240Z","title":"Zoology: Measuring and improving recall in efficient language models.arXiv preprint arXiv:2312.04927, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.494240Z"},"links":{"cited_paper":"/paper/2312.04927","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:795cf630f246b1db65f23d8f3fb8b605d8da28e9d85baa6aa3436abf561c0b4f","observation_id":"c6d43768-d201-4cbb-9be7-612c124966d4","resolution":{"observed_at":"2026-08-06T20:57:47.494240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.557374Z","title":"Copernicus, New York, NY , USA, 1996","venue":null,"work_id":null,"year":1996},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.557374Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c4bf2aa8b0e864cf53bb723d85b2242108301c17a90dbc1b3e7d411363d82509","observation_id":"f5ac2633-72d9-4809-ba23-b5ac0b410e82","resolution":{"observed_at":"2026-08-06T20:57:47.557374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.637617Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.Transactions on Machine Learning Research, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.637617Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:9b04a28f2d81fdf6602ff6e7cdb9ef84a0e6f3a490fc3e17d5bb960f3e342f1e","observation_id":"207607ea-25dc-4828-8ae7-4bb8ce2d546f","resolution":{"observed_at":"2026-08-06T20:57:47.637617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.729232Z","title":"Finding transformer circuits with edge pruning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.729232Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:51507141345ac68bf1609eb91b0a4feaae71ab33a5587fd684404c9a1eee8ce4","observation_id":"032bf900-263b-4253-9a51-6e01fb661c56","resolution":{"observed_at":"2026-08-06T20:57:47.729232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.830622Z","title":"Language models are few-shot learners.Advances in neural information processing systems, 33:1877–1901, 2020","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.830622Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:ee5fed9421b29e83f7949ed72c5b2ff0dc2879545a79fccc7e87de3508b2181c","observation_id":"78db4f16-32ca-47de-b375-2d4c0c16fbb0","resolution":{"observed_at":"2026-08-06T20:57:47.830622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.922775Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.922775Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:b6226c7b92cd86ac4f2245e299a9582d53c2fd83bb5f82cd4a1f2fd8f5fc9ea7","observation_id":"db2150c0-0f3e-41b5-8df8-7b6ffbd542bb","resolution":{"observed_at":"2026-08-06T20:57:47.922775Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:47.990675Z","title":"Toward understanding in-context vs","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:47.990675Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:8fd400b8392e2d5c78598694fb52dd47a1e3b1fb63a1172151cdccee3ad5b026","observation_id":"8f2873f2-e3ae-4c09-a79f-3e64fb34b23e","resolution":{"observed_at":"2026-08-06T20:57:47.990675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.058830Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.058830Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:84f4746daedfd6e3e1f4337562f42a944698ec3168b40526844d228263e35a77","observation_id":"2b8c0232-67ae-42da-9f40-579c5dce449a","resolution":{"observed_at":"2026-08-06T20:57:48.058830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.125248Z","title":"Sudden drops in the loss: Syntax acquisition, phase transitions, and simplicity bias in MLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.125248Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:073287e3ff8abf99f11da2990514b84f7548de6c6f3e29e77538bc4e31e6d3a8","observation_id":"155da576-b9f3-48bd-9e5f-055be0e4a8a8","resolution":{"observed_at":"2026-08-06T20:57:48.125248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.684738Z","title":"Quantifying semantic emergence in language models, 2024","venue":null,"work_id":"ba5c0422-dce1-426c-92e2-6b89604b982c","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.199456Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:8a0b5b0c7fa06807773bdb00992b4a905410f7c88f8a9def461b60a53310b8e1","observation_id":"b051e9bc-f5c2-4236-8f43-17bc5f1db24e","resolution":{"observed_at":"2026-08-06T20:57:51.689610Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.670424Z","title":"Dynamical versus bayesian phase transitions in a toy model of superposition, 2023","venue":null,"work_id":"7514f4fd-bde6-46bd-ac3e-83895712c48e","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.285810Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:bde2baf8d3dc3a1468aa4ed9f919a9ffe948527f65f0424db3596e0353cd17c8","observation_id":"bc162cb3-65db-437b-a29f-d57d543a9333","resolution":{"observed_at":"2026-08-06T20:57:51.675192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.06173","last_updated":"2023-03-10T19:16:53Z","snapshot_observed_at":"2026-07-06T15:01:30.019542Z","submitted_at":"2023-03-10T19:16:53Z","title":"Unifying Grokking and Double Descent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.06173","snapshot_observed_at":"2026-08-06T20:57:48.356869Z","title":"Unifying grokking and double descent","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.356869Z"},"links":{"cited_paper":"/paper/2303.06173","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fda61d868e94ffa0212b70db46494d7b8d3c420420897831d4d73c2eefdb3c5d","observation_id":"9ace98a6-1810-4e4e-b52f-24edb86128f2","resolution":{"observed_at":"2026-08-06T20:57:48.356869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.655863Z","title":"Can transformers learn optimal filtering for unknown systems?IEEE Control Systems Letters, 7:3525–3530, 2023","venue":null,"work_id":"0bac690a-ba8f-4ce3-a866-d7d60046824b","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.421421Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:95b7df41e36c58ccbe458ae0f3f7f4b5be3b12477020a0e865892ad7fa5ae960","observation_id":"36919971-e03f-4af6-87ec-6f6dbf4a9630","resolution":{"observed_at":"2026-08-06T20:57:51.660506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.640289Z","title":"Understanding emergent abilities of language models from the loss perspective, 2025","venue":null,"work_id":"6ed6f7e8-416b-4892-9cc2-19189c6dd57d","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.489563Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:86e4fffb339133ef633fc008ba7712eb1f2bd1ecbad40b785ab619bcb0e8f83c","observation_id":"d5873c6d-c0ee-4533-a4bd-ff4e6cce04e3","resolution":{"observed_at":"2026-08-06T20:57:51.645956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11004","last_updated":"2024-02-16T18:28:36Z","snapshot_observed_at":"2026-08-03T19:49:17.673100Z","submitted_at":"2024-02-16T18:28:36Z","title":"The Evolution of Statistical Induction Heads: In-Context Learning Markov Chains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11004","snapshot_observed_at":"2026-08-06T20:57:48.557237Z","title":"The evolution of statistical induction heads: In-context learning markov chains.arXiv preprint arXiv:2402.11004, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.557237Z"},"links":{"cited_paper":"/paper/2402.11004","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d0732f9c3a7891632516f1cc14c3a90a8d1ce17794beb53b0087ddd7e03871b1","observation_id":"6f0fa652-2942-45fe-a97e-2280ce290262","resolution":{"observed_at":"2026-08-06T20:57:48.557237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.634788Z","title":"A mathematical framework for transformer circuits.Transformer Circuits Thread,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.634788Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:33b5d97fdef5455abab57b235614f9988d34399534a50bce1518944d8f723372","observation_id":"ee229cd8-fb24-4b55-9fe6-b5c160e377e9","resolution":{"observed_at":"2026-08-06T20:57:48.634788Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.606042Z","title":"Predictability and surprise in large generative models","venue":null,"work_id":"902687e1-e391-4d3a-acab-5d032c04d6d3","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.802653Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:175e85e210ec710cfdf25f6e180a3f9c49b7a60af5e31cd01009f609fb0dd870","observation_id":"d7461706-7e55-4ca8-9e43-a9e4ee485eb2","resolution":{"observed_at":"2026-08-06T20:57:51.611107Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.874830Z","title":"What can transformers learn in-context? a case study of simple function classes.Advances in Neural Information Processing Systems, 35:30583–30598, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.874830Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:ce20626e77e19f88c527c0088651baa886b9c400439aacf06e4ba07b693d0f12","observation_id":"f523d299-edc4-48a9-ade2-efcce5a27bf7","resolution":{"observed_at":"2026-08-06T20:57:48.874830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.00838","last_updated":"2024-06-07T21:59:52Z","snapshot_observed_at":"2026-07-06T17:23:51.547578Z","submitted_at":"2024-02-01T18:28:55Z","title":"OLMo: Accelerating the Science of Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.00838","snapshot_observed_at":"2026-08-06T20:57:48.954246Z","title":"Olmo: Accelerating the science of language models.arXiv preprint arXiv:2402.00838, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.954246Z"},"links":{"cited_paper":"/paper/2402.00838","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:a64b538a81d8794fe4e2920ea782f9bc15f868a3a332eb2d6ca0450dba623b52","observation_id":"33e27c69-de1b-4d9d-81d4-70126a5b76c8","resolution":{"observed_at":"2026-08-06T20:57:48.954246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.02679","last_updated":"2023-01-06T19:00:01Z","snapshot_observed_at":"2026-08-06T08:20:11.326697Z","submitted_at":"2023-01-06T19:00:01Z","title":"Grokking modular arithmetic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.02679","snapshot_observed_at":"2026-08-06T20:57:49.031244Z","title":"Grokking modular arithmetic.arXiv preprint arXiv:2301.02679, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.031244Z"},"links":{"cited_paper":"/paper/2301.02679","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:345d2c9e340fea5bcd8de6a9c03f90aa77f0c36cdc284f65401d33679e175a79","observation_id":"c11acdc7-b915-4c30-bbee-d21ed541daf4","resolution":{"observed_at":"2026-08-06T20:57:49.031244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.580758Z","title":"Loss landscape degeneracy drives stagewise development in transformers, 2025","venue":null,"work_id":"af880b6b-834b-4016-ab86-6a74d41f26c2","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.103661Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fb9dab24fa41cf05091ebb5ab3e19d0e525355a7c947b9b15fea1f28ef6d9c86","observation_id":"7d6cc3ad-fdba-435e-bbe0-8ec343865c1a","resolution":{"observed_at":"2026-08-06T20:57:51.586092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:49.187011Z","title":"Neural networks and physical systems with emergent collective computational abilities.Proceedings of the national academy of sciences, 79(8):2554–2558, 1982","venue":null,"work_id":null,"year":1982},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.187011Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:374089135991eba30f9df151ba25fe5487f6446b98a9030874709723aec52017","observation_id":"fd601a35-8b4c-4b53-a661-43ef27000f66","resolution":{"observed_at":"2026-08-06T20:57:49.187011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.554727Z","title":"Task descriptors help transformers learn linear models in-context","venue":null,"work_id":"e4fcb9cf-c98c-4c47-86b7-8cf3c285cd6c","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.284731Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c80da38acf6c70ee1bd0282b3c01e1b6018c49124158fba42db2af2fa4362fc6","observation_id":"5fbf382b-11c6-4e9f-bcfa-8a1bcefa9927","resolution":{"observed_at":"2026-08-06T20:57:51.560528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.540443Z","title":"Deep networks always grok and here is why","venue":null,"work_id":"eca72e53-ccbf-47c1-a044-19b1f8d95ff7","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.292777Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:4242880d7a1f13338182db9e3db735273737680d13d02e9dd6cecd75c32b5efb","observation_id":"7ff2b510-b71f-43d7-94ff-6aa2798a8dae","resolution":{"observed_at":"2026-08-06T20:57:51.544878Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.525797Z","title":null,"venue":null,"work_id":"cb7dd801-a1c4-4aae-aaaf-9cd75393edab","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.386295Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:33ff9650f7f1aff24b851948988253d4be0dbab2d0c8f0e907d1a10dff1161eb","observation_id":"67e4b03f-617a-4dc8-b0a4-ce4bd50f7f8e","resolution":{"observed_at":"2026-08-06T20:57:51.530373Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.510425Z","title":"Grokking as the transition from lazy to rich training dynamics","venue":null,"work_id":"69de77e9-f44a-4018-bd5f-f64898ccde0e","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.521069Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:2c17e6cc4c3837b6b5649a6ce6e828ad6b469e20d4561cb8c474ef698503e2cd","observation_id":"5780ac73-d04a-4e0c-b3b4-8316f3714c72","resolution":{"observed_at":"2026-08-06T20:57:51.515532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.494514Z","title":null,"venue":null,"work_id":"985bb60a-a72b-4228-9901-5c2d9e174dbe","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.640425Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e9fff37ff219d9ccabd788d9e8434cde093d2311670e090810e94f6d38c169f6","observation_id":"1aa3af2f-3150-4b30-a59a-66594e8a73f9","resolution":{"observed_at":"2026-08-06T20:57:51.499994Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.480240Z","title":"The local learning coefficient: A singularity-aware complexity measure, 2024","venue":null,"work_id":"92a18f35-c9cb-4658-9448-8c798d51dc32","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.772445Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c6c0c6162476dfadd095124c5a46f59e4f6c28e14a42b548d412c8724473bec1","observation_id":"64c6aff3-b30d-44fe-abd8-6fadb4733a06","resolution":{"observed_at":"2026-08-06T20:57:51.484652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14578","last_updated":"2024-10-28T04:10:40Z","snapshot_observed_at":"2026-07-06T18:18:39.093732Z","submitted_at":"2024-05-23T13:52:36Z","title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","version":5},"cited_work":{"arxiv_id":"2405.14578","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.14578","snapshot_observed_at":"2026-08-06T20:57:50.724970Z","title":"Surge Phenomenon in Optimal Learning Rate and Batch Size Scaling","venue":"cs.LG","work_id":"289492d8-e0ad-4faa-b970-1cad93a5e2e1","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:49.881070Z"},"links":{"cited_paper":"/paper/2405.14578","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:cda9f9d8042747411e3082695850bbcd0fa5c1efd02eb8eb2ab8d997a4777441","observation_id":"cca60b8a-8980-4b77-abc3-e7d778eb3db3","resolution":{"observed_at":"2026-08-06T20:57:50.729788Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.465792Z","title":"Trans- formers as algorithms: Generalization and stability in in-context learning","venue":null,"work_id":"1b0ddba0-ca96-4a0b-8ee1-311e36417258","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.007600Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:50e771c43b593068e4229f11a1825ea2a6ce2d38344e8e85d8cd1ec6afb0742e","observation_id":"f77a2e13-8aeb-4532-8a80-ea132a00d750","resolution":{"observed_at":"2026-08-06T20:57:51.470598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.451384Z","title":"Dual operating modes of in-context learning, 2024","venue":null,"work_id":"d9e85277-b706-4175-aad9-aaed57e046f3","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.124977Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d728c44ceafb26a61176b401601e15726b673e956464734dda05234a74fdf572","observation_id":"99bfa400-d578-497d-bea8-ea8c3d4b4f5e","resolution":{"observed_at":"2026-08-06T20:57:51.456102Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.437420Z","title":"Can transformers solve least squares to high precision? InICML 2024 Workshop on In-Context Learning, 2024","venue":null,"work_id":"ce89b72d-0fad-4e5f-9870-79547f729e94","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.145177Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c699ce221f6600b330e32f26b2327c96ed90d794a30c3fb19c8d7fd5c943367b","observation_id":"f38d062f-5859-480d-8a95-86e6c878fafc","resolution":{"observed_at":"2026-08-06T20:57:51.441982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.201250Z","title":"Liu, Kevin Lin, John Hewitt, Ashwin Paranjape, Michele Bevilacqua, Fabio Petroni, and Percy Liang","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.201250Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:9dc4383c62ee857c653d36a5adc444c1b62d4e665dd7d65a53b031f8b12386d3","observation_id":"7b828d0f-6eb3-461a-b0e5-a0942822f9e1","resolution":{"observed_at":"2026-08-06T20:57:50.201250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.414500Z","title":"Omnigrok: Grokking beyond algorithmic data","venue":null,"work_id":"4eb67290-712a-4fd0-adf5-0a234bf1a77e","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.316022Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:86504fb640dbabf53ba5e6d6c5feb79936c5e98b1ba184887e4c30a6dcc0b61c","observation_id":"671c5f59-d0c5-47c4-ba43-fc4270979382","resolution":{"observed_at":"2026-08-06T20:57:51.418824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.327309Z","title":"Decoupled weight decay regularization, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.327309Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:65591e71c7c3c6ca3503d8ef9f349cf9832514d0e3b2e6c71a99548a3ccc7d7d","observation_id":"2688f4e1-93c3-4562-ba48-a348e8927a75","resolution":{"observed_at":"2026-08-06T20:57:50.327309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.01809","last_updated":"2024-07-15T12:21:56Z","snapshot_observed_at":"2026-08-05T15:26:10.962792Z","submitted_at":"2023-09-04T20:54:11Z","title":"Are Emergent Abilities in Large Language Models just In-Context Learning?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.01809","snapshot_observed_at":"2026-08-06T20:57:50.362140Z","title":"Are emergent abilities in large language models just in-context learning?arXiv preprint arXiv:2309.01809, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.362140Z"},"links":{"cited_paper":"/paper/2309.01809","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d6b978118eb662a9f6a0240142b5852c20cebd89f455d4cd116a02536e3d2311","observation_id":"2e1679c7-fd4d-487e-9ea8-b2e1e801596d","resolution":{"observed_at":"2026-08-06T20:57:50.362140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.389966Z","title":"Dick, and Hidenori Tanaka","venue":null,"work_id":"3c082550-ce4c-452c-92b2-5706ab63e393","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.366585Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:567d54e6bd4faf8176870b8531af38388db70c67d5e129bb3d30307ce6da18db","observation_id":"0ddb44b2-94e3-4b9f-b5d3-04c049609a5a","resolution":{"observed_at":"2026-08-06T20:57:51.394536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.18817","last_updated":"2024-04-02T05:43:18Z","snapshot_observed_at":"2026-07-06T16:55:13.127567Z","submitted_at":"2023-11-30T18:55:38Z","title":"Dichotomy of Early and Late Phase Implicit Biases Can Provably Induce Grokking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.18817","snapshot_observed_at":"2026-08-06T20:57:50.370208Z","title":"Dichotomy of early and late phase implicit biases can provably induce grokking.arXiv preprint arXiv:2311.18817, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.370208Z"},"links":{"cited_paper":"/paper/2311.18817","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:7a42943ffcb3dd7a2ccf37a8bfa7d67aa7dc6966832243cb055ce9b97a821b31","observation_id":"af9cca3c-f0ec-4dc1-9f81-3963a84a4100","resolution":{"observed_at":"2026-08-06T20:57:50.370208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.375536Z","title":"A practical bayesian framework for backpropagation networks.Neural computation, 4(3):448–472, 1992","venue":null,"work_id":"2639d473-5147-40b2-87a4-d090a3bbb637","year":1992},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.374733Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c854c906df08945965e75df6595d0b15100d7aff774a9306175502a42b8512b7","observation_id":"bb2a53ad-8667-469a-98be-b25573be7a5c","resolution":{"observed_at":"2026-08-06T20:57:51.380244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.361220Z","title":"Exact learning dynamics of in-context learning in linear transformers and its application to non-linear transformers, 2025","venue":null,"work_id":"0328eb25-74df-497c-8eec-06fe82dd4b17","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.379494Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:9de810cd75b3c97e1738b060d22929bad5a727bacb330e08b730975b360ba70e","observation_id":"0f1e8a9f-4ec7-4a60-9e96-f23fc3968f19","resolution":{"observed_at":"2026-08-06T20:57:51.365506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.347821Z","title":"Emergence in non-neural models: grokking modular arithmetic via average gradient outer product","venue":null,"work_id":"a2885b51-fc6b-4dbb-a411-b2206d5303d1","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.383332Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:492567b9037b6398baa17f0c196a405e4a0a0667e6fb103fbcfa4d44d5f168b4","observation_id":"b811f6a1-233f-4b67-9aa6-3e190e35d328","resolution":{"observed_at":"2026-08-06T20:57:51.352256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.334516Z","title":"Hoffman, and David M","venue":null,"work_id":"a88e7619-d1f0-46b4-b5bb-46a3d9f7927c","year":2017},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.387426Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:a1ac6e0c7a0c1a855f138e511c62d20aa7fdcc2f76105bdb0b60ed847e2023c2","observation_id":"2e476f30-2f64-4823-8ab4-256fca7fa2b9","resolution":{"observed_at":"2026-08-06T20:57:51.338471Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"math-ph/0609050","last_updated":"2007-02-27T14:22:05Z","snapshot_observed_at":"2026-07-07T06:16:50.870997Z","submitted_at":"2006-09-18T11:18:42Z","title":"How to generate random matrices from the classical compact groups","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"math-ph/0609050","snapshot_observed_at":"2026-08-06T20:57:50.391686Z","title":"How to generate random matrices from the classical compact groups","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.391686Z"},"links":{"cited_paper":"/paper/math-ph/0609050","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:114324908808791c0938718d54e3ae02017c93044dc12b1712e0e725818d1022","observation_id":"89ea9f58-4226-4de9-8aa3-64b1b117e973","resolution":{"observed_at":"2026-08-06T20:57:50.391686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.396097Z","title":"The quantization model of neural scaling.Advances in Neural Information Processing Systems, 36, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.396097Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c5eaeb6986b39e4fce6df09671d9a374965553922b32b02743bea8ca5eafd3d5","observation_id":"000462f9-ae81-4b07-b946-d478a9b70acd","resolution":{"observed_at":"2026-08-06T20:57:50.396097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.310350Z","title":"Rethinking the role of demonstrations: What makes in-context learning work?, 2022","venue":null,"work_id":"7c4291fe-f271-452c-a5d5-ff751d017625","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.399865Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:31267d158364d4877d9b1670a4127c10708fc1e7a3517ca6448bd72b332dccca","observation_id":"b818bee2-639b-4d05-9836-953ecb2326b9","resolution":{"observed_at":"2026-08-06T20:57:51.314342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.296414Z","title":null,"venue":null,"work_id":"21ca1026-ab0d-475c-bd41-7fc81cbe9608","year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.404218Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:99aa0bcbe50ae0ad5fa4d371c306379d843b1e96474323318e4eee43110abe63","observation_id":"de8c6695-0166-46fa-bb2c-b0af4bfa9530","resolution":{"observed_at":"2026-08-06T20:57:51.301004Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.282471Z","title":"Grokking mod- ular arithmetic can be explained by margin maximization","venue":null,"work_id":"150f77f4-dfe0-40b8-84f3-efa89f6f72f9","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.408153Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:6c58c809f9c8a6da926b5e6543931977bf000d05169b228eee91710c0681c293","observation_id":"12f57399-d789-4112-9fbb-12097e4db8c2","resolution":{"observed_at":"2026-08-06T20:57:51.286523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.269207Z","title":"Transformers can do bayesian inference, 2024","venue":null,"work_id":"dfcdae16-1906-4cec-84ca-9d14c2e56398","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.412570Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:ab31b2c737b95a11612dba372fb933451c53d4d991237080a91f48db4ad036d1","observation_id":"c4136913-72b4-4aa7-80dc-acff20b83f9c","resolution":{"observed_at":"2026-08-06T20:57:51.272974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.256445Z","title":null,"venue":null,"work_id":"1065df3b-100b-40f1-b7b3-af919221b570","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.416531Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:b2fcd3530032e268bc693c819e56e42240d91fef3f051eaf8aeb532d3a508466","observation_id":"4afa61df-3135-4f40-9fbe-3d2fc9b9d5c5","resolution":{"observed_at":"2026-08-06T20:57:51.260246Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.242918Z","title":"Progress measures for grokking via mechanistic interpretability","venue":null,"work_id":"c2d516b5-5a03-4c3f-9036-fc2ca2ac4533","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.420860Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:8f3e4389dfef180e3f01de56764b229596b6428a21613877590709f75687ecd8","observation_id":"39032d8a-d2b7-44af-8ad3-b2e3b7d80178","resolution":{"observed_at":"2026-08-06T20:57:51.247422Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.229725Z","title":"Differential learning kinetics govern the transition from memorization to generalization during in-context learning, 2024","venue":null,"work_id":"42207c71-492e-41e5-aa93-e82844e48d27","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.424811Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:0cfc47be0fc2de87e66925b28ddbbab79cb11f03a751f03e18898783d37dcccc","observation_id":"5cba795c-5d46-425e-8f06-e0ed8037d815","resolution":{"observed_at":"2026-08-06T20:57:51.233720Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.428875Z","title":"Lee, and Alberto Bietti","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.428875Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d25ecd2aed5848d44daa3a76daf4beaafb54f84aadfa6677874502f33c5358f7","observation_id":"dfae44a1-5038-4732-a3fd-6f2bae8a87f4","resolution":{"observed_at":"2026-08-06T20:57:50.428875Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.206879Z","title":null,"venue":null,"work_id":"74916ea7-7b3c-449b-9b30-f862410f9ef9","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.433106Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:00a7aebdb14eed0ff7e2dff46d674bd7903e94181c135fe6bda6011cbb868793","observation_id":"fc02881c-073f-4447-a58a-c6529843e325","resolution":{"observed_at":"2026-08-06T20:57:51.210591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-06T20:57:50.436894Z","title":"In-context learning and induction heads.arXiv preprint arXiv:2209.11895, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.436894Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:bfaf7fe8e8253f076dad82e59f5890f83a9828b76f0d5ed569b2816e41f70c35","observation_id":"723e21ed-5bbb-494a-ba82-8819d30f6d07","resolution":{"observed_at":"2026-08-06T20:57:50.436894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.194197Z","title":"What in-context learning \"learns\" in-context: Disentangling task recognition and task learning, 2023","venue":null,"work_id":"54a43f14-70c0-4ba3-aebf-0a4e62f7b8b7","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.440839Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:be2cba571e207f7cfbc8aa771a5b1a5f2e19000a69ea6fdba3effe9d6d03a631","observation_id":"804ec645-8b21-45a2-8494-2d2f930f556e","resolution":{"observed_at":"2026-08-06T20:57:51.198025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.182016Z","title":"In-context learning through the bayesian prism, 2024","venue":null,"work_id":"e5a4f42f-1cb0-4e8d-870e-17c2f76c8c00","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.444624Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:157fdfdd062e57974cc0dde619adecb5d42b8277abafb39eb89b42906480d8b6","observation_id":"a31572cc-2f52-44f7-99eb-3f535e10d3d3","resolution":{"observed_at":"2026-08-06T20:57:51.185916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.169128Z","title":"Competition dynamics shape algorithmic phases of in-context learning, 2025","venue":null,"work_id":"ab6369be-afa6-4150-b631-f8f7994c617c","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.448214Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:2dfe73da92e220dd60a37a46e3129fe433b72a3da9aac723989580e2f87c5a32","observation_id":"30793b95-aaa8-4e66-ab42-9d3ac3c85cb2","resolution":{"observed_at":"2026-08-06T20:57:51.173103Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.155984Z","title":"Gradient starvation: A learning proclivity in neural networks.Advances in Neural Information Processing Systems, 34:1256–1272, 2021","venue":null,"work_id":"5d2366f3-8561-4dd6-95a6-873faebf148a","year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.451939Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:00862b45e4aefee4449c60087917579274b1c41fe4fd882f4a6829d91a3059df","observation_id":"56e50786-2ff2-4886-9bf7-18a2bb1b7030","resolution":{"observed_at":"2026-08-06T20:57:51.160478Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.141974Z","title":"Multi-scale feature learning dynamics: Insights for double descent","venue":null,"work_id":"36231feb-343a-444f-9b9e-e4b95636ab9e","year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.456276Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:329602c07e516a8c8bfe871f76cfd4c61ba8f38f6617f655ab7d1f3b3b4b0106","observation_id":"1bd7c59a-1a78-4b51-b1df-733fd66ea5f2","resolution":{"observed_at":"2026-08-06T20:57:51.146280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.02177","last_updated":"2022-01-06T18:43:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-06T18:43:37Z","title":"Grokking: Generalization Beyond Overfitting on Small Algorithmic Datasets","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.02177","snapshot_observed_at":"2026-08-06T20:57:50.459920Z","title":"Grokking: Gen- eralization beyond overfitting on small algorithmic datasets.arXiv preprint arXiv:2201.02177, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.459920Z"},"links":{"cited_paper":"/paper/2201.02177","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:4362fd9d8dbd02f50921dee6ed012a7a10e2d68e5e1c6398de64c1630e685f00","observation_id":"0cfea553-1a99-4ff6-9f62-f81d01c3d2a3","resolution":{"observed_at":"2026-08-06T20:57:50.459920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04697","last_updated":"2025-05-19T11:02:53Z","snapshot_observed_at":"2026-08-05T11:14:36.679405Z","submitted_at":"2025-01-08T18:58:48Z","title":"Grokking at the Edge of Numerical Stability","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04697","snapshot_observed_at":"2026-08-06T20:57:50.463997Z","title":"Grokking at the edge of numerical stability.arXiv preprint arXiv:2501.04697, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.463997Z"},"links":{"cited_paper":"/paper/2501.04697","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:5a3c6364c044dd290ec957dba70ffe9583bcf33803819f9fb6862ad77c65f463","observation_id":"1333739d-8086-4ea3-b2b4-c1b15760c450","resolution":{"observed_at":"2026-08-06T20:57:50.463997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.128136Z","title":"Transformers on markov data: Constant depth suffices","venue":null,"work_id":"368e05cc-c36a-4d5a-b493-36eb8411505c","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.469390Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:a7512027a52a203ff232d3c4ec85eb4d1b5de9a4f4bc481b524754d877c0d027","observation_id":"0788a3e0-1ddb-4c08-8369-c2c6cdd85b21","resolution":{"observed_at":"2026-08-06T20:57:51.131965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.114868Z","title":"Pretraining task diversity and the emergence of non-Bayesian in-context learning for regression.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"41d3b2f6-4875-40fa-9761-7d5bfcd471b9","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.473168Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:2b7cb64e2a92b1a3223eec540bca6b2449a595609bdc18ead829ac1dfd7ca0cc","observation_id":"c138f444-e289-4a3a-955f-a983ca9ba18d","resolution":{"observed_at":"2026-08-06T20:57:51.119108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.101046Z","title":"The mechanistic basis of data dependence and abrupt learning in an in-context classification task, 2023","venue":null,"work_id":"dcf30d2f-b24e-44ef-87ac-c40921afb039","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.477710Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:88188c4618d51a5982e30b055a8bcc6fcb8d8ff108e784e42c0966ff989acd12","observation_id":"d8816f7d-d4f4-44ca-93a9-b11c684ea924","resolution":{"observed_at":"2026-08-06T20:57:51.105229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.088487Z","title":null,"venue":null,"work_id":"d4e546a9-17a5-4857-9487-d30838eb36fa","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.482042Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:478f00cebae471d9118bf0be484d41c25e7f7bbee0cf20b2fb5ec4d21964daef","observation_id":"3b117b78-3ca9-46d5-be1f-4c625d258c31","resolution":{"observed_at":"2026-08-06T20:57:51.092216Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.074514Z","title":"Are emergent abilities of large language models a mirage?Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"37025723-47d9-4ad8-be3f-567a11aa71cd","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.485665Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:db250e71da20a4d9d4058d3090d4b125b90a6b5b796e6dcbea799439e1864cad","observation_id":"d0be3b76-8c60-4dca-9dfc-0f69584c4c0b","resolution":{"observed_at":"2026-08-06T20:57:51.079229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.060407Z","title":"I preliminaries","venue":null,"work_id":"3e341b72-aeed-476f-85ac-60790694426b","year":1999},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.489660Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:d9031b11c91dfc9b978946c5423a61d8a62044c20814e048c610f21690270de0","observation_id":"c0b2b5b6-35a6-4236-92a8-1a58e2b0ade3","resolution":{"observed_at":"2026-08-06T20:57:51.064250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.046586Z","title":"The pitfalls of simplicity bias in neural networks.Advances in Neural Information Processing Systems, 33:9573–9585, 2020","venue":null,"work_id":"947596f0-975b-4a97-81ba-cfa2618bb024","year":2020},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.493585Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:e40117eaee5d24a5f07b8f3e9cf79c8e5f40a191484e07aeed0ba7994ad1fef1","observation_id":"242a903a-fc53-44cc-a146-c3c17f33ccab","resolution":{"observed_at":"2026-08-06T20:57:51.051341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.033317Z","title":"Singh, Ted Moskovitz, Sara Dragutinovic, Felix Hill, Stephanie C","venue":null,"work_id":"cc08632f-65e2-4260-9d1c-2b8420eb97c0","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.497528Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:bc6adccbc0273c046b3212ae4fa5268ef5a00712c2a462d48495807338b3ce58","observation_id":"9c6ec1f4-3189-431b-867b-8c21c43043e3","resolution":{"observed_at":"2026-08-06T20:57:51.037428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.020116Z","title":"Singh, Ted Moskovitz, Felix Hill, Stephanie C","venue":null,"work_id":"300aa5e7-b2f8-4055-9e58-212e49098bd4","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.500971Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:231416c46cca7cd16beb0c39aa00ea90f5b9a36a10468ec282084d35fdbf1bbd","observation_id":"2c1c3dee-e0cc-4819-bf45-53af02abf929","resolution":{"observed_at":"2026-08-06T20:57:51.024189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:51.006499Z","title":"The implicit bias of gradient descent on separable data.Journal of Machine Learning Research, 19:1–57, 2018","venue":null,"work_id":"0285c27e-a834-425d-b7be-2a90b70bcd7a","year":2018},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.504540Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:f632445d9a7913ae363a0ad2c03ffb19f9e4b86215146d3282442588a99eb7e7","observation_id":"a103b708-0a07-4d13-bc18-0ea7a8e788ec","resolution":{"observed_at":"2026-08-06T20:57:51.010579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.508348Z","title":"Transformers learn in-context by gradient descent","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.508348Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:c55df11058de8b5930a6070c8f8b53d902cd2cb4c5b51f8b12ad2a9b22ac4c63","observation_id":"c891c0cf-7476-4b69-955f-2c8bae446104","resolution":{"observed_at":"2026-08-06T20:57:50.508348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.984343Z","title":"Label words are anchors: An information flow perspective for understanding in-context learning, 2023","venue":null,"work_id":"53975480-460a-4e01-a840-77afcbe2fd92","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.512389Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:224fa934b42b2176d60bc6a0efb219defe311beb8179f6648f0a5aaede7a2530","observation_id":"dfe94556-f4bf-4227-9d1b-a21d3519d5a1","resolution":{"observed_at":"2026-08-06T20:57:50.988282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.971609Z","title":"Investigating the pre-training dynamics of in-context learning: Task recognition vs","venue":null,"work_id":"edfae7be-2d3b-4707-a56a-505a3c5a9993","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.516098Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:8b368c25879e7de3aaf0724105929dc9836a67a5cdbe0e771489eb81b13bd6a2","observation_id":"7d418978-8f03-473f-8a2a-27b221e519d0","resolution":{"observed_at":"2026-08-06T20:57:50.975604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.958201Z","title":"Cambridge monographs on applied and computational mathematics ; 25","venue":null,"work_id":"1db6a00c-7334-46b2-922b-3fe85a9c827a","year":2009},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.519925Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:1782e7f33b4c00b48ef8c0350b543e799ac1fc2e32a12f36053b94cd43491486","observation_id":"31f1e21a-5422-4bd8-9538-0f6d41977210","resolution":{"observed_at":"2026-08-06T20:57:50.962394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.523610Z","title":"Emergent abilities of large language models.Transactions on Machine Learning Research, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.523610Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:4cea4593bd9ada92d20245a72db122c98a69226930c181718f35c16bb3e61016","observation_id":"7f2b0d05-542c-44dd-9d49-b6bb9ca53394","resolution":{"observed_at":"2026-08-06T20:57:50.523610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.934459Z","title":"Symbol tuning improves in-context learning in language models","venue":null,"work_id":"dcb89470-3ab5-4954-8a70-dcfcc67fa2bd","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.527132Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:3e7da8e0456741ed98c6f2eac33187dd40efdf36e4f8f3082ac2e6de0c046b46","observation_id":"89e7c1c8-f9d5-48f1-8ca4-3c063d7021c9","resolution":{"observed_at":"2026-08-06T20:57:50.938971Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.920072Z","title":"Larger language models do in-context learning differently, 2023","venue":null,"work_id":"356f1526-1c4a-4fbe-ab02-cc878185a323","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.530600Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:fe2708c7c0615be29548bba697273a90c3f3a26424c9fd73af5d1cb3f86b5a99","observation_id":"c235e701-5b2f-423d-9b9b-ab38e5b64c8a","resolution":{"observed_at":"2026-08-06T20:57:50.924799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.904822Z","title":"The learnability of in-context learning, 2023","venue":null,"work_id":"cabbd6c7-45e6-4877-82b8-2307d983d8ff","year":2023},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.534207Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:2ca50e2113f4fa09ab3e9e7ff4509471c8c5c4a8f76cde4742db54b2501479d6","observation_id":"4f03d646-60e7-4d00-bb02-6a3de0a0c218","resolution":{"observed_at":"2026-08-06T20:57:50.909234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.889339Z","title":"Bartlett","venue":null,"work_id":"13d59e70-5e81-41ae-946f-aa42071ea961","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.538652Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:8f62c854f4579de31020ab218c8775144be09a989184273847217cecf56d2685","observation_id":"443613bf-e32b-4658-8855-fc8e5a97ae2c","resolution":{"observed_at":"2026-08-06T20:57:50.893332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02080","last_updated":"2022-07-21T07:44:13Z","snapshot_observed_at":"2026-07-30T03:45:35.558793Z","submitted_at":"2021-11-03T09:12:33Z","title":"An Explanation of In-context Learning as Implicit Bayesian Inference","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.02080","snapshot_observed_at":"2026-08-06T20:57:50.542232Z","title":"An explanation of in-context learning as implicit bayesian inference.arXiv preprint arXiv:2111.02080, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.542232Z"},"links":{"cited_paper":"/paper/2111.02080","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:b5074cd30292677b7ec275a0c43b470f0eea527d020d6ab29fc08eb0b1cc25bb","observation_id":"9aa4c803-6d33-4dd4-81a1-bd1920753f32","resolution":{"observed_at":"2026-08-06T20:57:50.542232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.875523Z","title":"Which attention heads matter for in-context learning?, 2025","venue":null,"work_id":"9e7dd0ff-df84-4d02-ac18-c18c3ed991e4","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.545964Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:b733f971d8ccd362b349fa4879bb50838e10e6fcef5dc99db893416ebba3f44e","observation_id":"8626323f-a16e-49be-9356-71764c22750d","resolution":{"observed_at":"2026-08-06T20:57:50.879642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1709.06493","last_updated":"2017-10-03T14:31:03Z","snapshot_observed_at":"2026-07-06T06:00:20.967053Z","submitted_at":"2017-09-19T15:55:16Z","title":"Learning to update Auto-associative Memory in Recurrent Neural Networks for Improving Sequence Memorization","version":3},"cited_work":{"arxiv_id":"1709.06493","doi":null,"metadata_source":"pith","pith_arxiv_id":"1709.06493","snapshot_observed_at":"2026-08-06T20:57:50.601484Z","title":"Learning to update Auto-associative Memory in Recurrent Neural Networks for Improving Sequence Memorization","venue":"cs.AI","work_id":"b15eb500-f2a4-4eb6-b082-a5fb3c8f3148","year":2017},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.549945Z"},"links":{"cited_paper":"/paper/1709.06493","citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:88d46a134e3588ec96dc6792633848b3b0b254b7f06690d73f8e8b63b5e86702","observation_id":"ed80c7c0-bf72-4ce1-9cc5-baa668260520","resolution":{"observed_at":"2026-08-06T20:57:50.607902Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.860908Z","title":"Singh, Peter E","venue":null,"work_id":"94ecc74d-e7f9-48c6-94f7-c2faa5e15ec3","year":2025},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.554227Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:bc91a53a4ca075fede960efc2946f8f358d1e0d87134c4ecf84eca9d35a154aa","observation_id":"c3428b09-90c6-4d74-81e6-aecdb2189132","resolution":{"observed_at":"2026-08-06T20:57:50.865767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.845945Z","title":"grokking","venue":null,"work_id":"698edad1-e1b9-4bb7-8ade-ea51d00fc6d9","year":2024},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.558118Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:f5f992929e7e1e9c0aa18a83b1f533264cdb6202a129209f1c494b98345d724d","observation_id":"e23d4f56-2c7b-436e-a12d-ccf86e2ee14d","resolution":{"observed_at":"2026-08-06T20:57:50.850211Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:50.832349Z","title":"tiny”, “small","venue":null,"work_id":"48043fc5-67f6-43a2-8b12-de967723f3fa","year":null},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:50.562642Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:0b33e6c4fec2e639349513fd1a27beadf515c4e4471b0511b78994a3eee466b4","observation_id":"7cf24358-5e8d-4b23-a0b3-ad11780b5920","resolution":{"observed_at":"2026-08-06T20:57:50.836438Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:57:48.723869Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall","version":2},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-06T20:57:48.723869Z"},"links":{"citing_paper":"/paper/2507.01414"},"observation_digest":"sha256:50e43313482e41295edb665874ce52574f7334d5374fae2bffb1cc8fb0f9ed65","observation_id":"1d1e9eaf-2462-474c-ba00-9998e195de43","resolution":{"observed_at":"2026-08-06T20:57:48.723869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.01414","last_updated":"2026-06-16T22:29:23Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T20:49:36.796746Z","submitted_at":"2025-07-02T07:09:09Z","title":"Decomposing Prediction Mechanisms for In-Context Recall"},"reference_resolution":{"displayed":89,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":2,"verified_fuzzy":47},"total_outbound_references":89},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 89 of 89 outbound references and 0 inbound Pith citation observations for arXiv:2507.01414."}