{"as_of":"2026-08-13T07:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0d5be04b81a441b514536c80be85b696963da176f7d8f34d68f36eee93f0289e","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T15:34:02.796850Z","state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2604.12426/citation-record","integrity":"/paper/2604.12426/integrity","json":"/paper/2604.12426/citation-record.json","paper":"/paper/2604.12426"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08112","last_updated":"2025-11-11T01:13:28Z","snapshot_observed_at":"2026-08-12T19:18:05.160367Z","submitted_at":"2023-03-14T17:47:09Z","title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","version":6},"cited_work":{"arxiv_id":"2303.08112","doi":"10.48550/arxiv.2303.08112","metadata_source":"pith","pith_arxiv_id":"2303.08112","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","venue":"cs.LG","work_id":"a127314f-7424-488f-b6d7-8214650c420f","year":2023},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2303.08112","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:e4e7e0a1d962840860e25ff9f56a3c9fea602ebb9397d2ce8f758191f2c83f45","observation_id":"ce70fbc0-6252-4ac6-be41-790364502c07","resolution":{"observed_at":"2026-05-12T16:54:37.831311Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-06-01T22:57:41.570848+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T22:57:41.570848+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.09673","last_updated":"2024-09-20T21:21:56Z","snapshot_observed_at":"2026-08-13T00:06:55.915196Z","submitted_at":"2024-05-15T19:27:45Z","title":"LoRA Learns Less and Forgets Less","version":2},"cited_work":{"arxiv_id":"2405.09673","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2405.09673","snapshot_observed_at":"2026-07-04T19:30:07.259186Z","title":"Biderman, J","venue":null,"work_id":"ca00d37c-5297-4fa7-b692-52260eb614f2","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2405.09673","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:ce31f3059e90dd9477cbe0c632a8c66fe962bd73c13f3a25d7cc700a2b91fa0a","observation_id":"be5aafc9-b8d7-4015-a724-00a650b1d822","resolution":{"observed_at":"2026-05-11T10:16:07.435143Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T14:57:14.742120Z","title":"Language models are few-shot learners.Advances in neural information processing systems, 33:1877–1901","venue":null,"work_id":"84780adb-76cf-44ac-a8b7-e24d4fa5c592","year":1901},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:b0ecf73616be3a9d41fbb99ace72ffd674503b427cfe823c6e5657fb0033bd48","observation_id":"827c7083-46af-43f0-95fc-48bc0aa77d24","resolution":{"observed_at":"2026-05-17T20:02:06.052943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.13898","doi":"10.48550/arxiv.2505.13898","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InAdvances in Neural Infor- mation Processing Systems, volume 36, pages 16318– 16352","venue":"ArXiv.org","work_id":"38df74c9-31e0-41ae-99bd-f00c7c1dd0c8","year":2025},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:b18b031a52e3c84dbe06cfe7ea5cbb5852e4b2a7361171188fea84e9bd8b4c25","observation_id":"826809d5-cfbe-4d39-8b51-dd857d0c8c14","resolution":{"observed_at":"2026-05-11T10:16:07.448865Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03819","last_updated":"2019-03-05T16:46:19Z","snapshot_observed_at":"2026-08-13T07:34:17.550838Z","submitted_at":"2018-07-10T18:39:15Z","title":"Universal Transformers","version":3},"cited_work":{"arxiv_id":"1807.03819","doi":"10.48550/arxiv.1807.03819","metadata_source":"pith","pith_arxiv_id":"1807.03819","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Universal Transformers","venue":"cs.CL","work_id":"8e5baefe-d209-411c-aefc-5acaa9275c8a","year":2018},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/1807.03819","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:c66b74f1c492f1ef63b20c5e390e765e92c686dd7cd76e9a5ddb5e43f359e942","observation_id":"0f598c69-f104-4c4f-aab3-1b5aac7e2269","resolution":{"observed_at":"2026-05-13T08:04:31.534239Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.15647","last_updated":"2025-05-12T03:51:20Z","snapshot_observed_at":"2026-08-12T22:39:05.389967Z","submitted_at":"2024-09-24T01:21:17Z","title":"Looped Transformers for Length Generalization","version":5},"cited_work":{"arxiv_id":"2409.15647","doi":"10.48550/arxiv.2409.15647","metadata_source":"arxiv_reference","pith_arxiv_id":"2409.15647","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2409.15647 , year=","venue":"arXiv (Cornell University)","work_id":"004f7149-3198-47ac-ac43-7a5b435a7368","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2409.15647","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:deb4011bf53d26b6f8ca2c4cd9b766eed6f962a8f3c141236404cd7f0dfa1b71","observation_id":"441218a9-2ed3-4888-a8ae-b08199911a77","resolution":{"observed_at":"2026-05-11T10:16:07.428196Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14561","last_updated":"2025-04-01T16:04:53Z","snapshot_observed_at":"2026-08-12T23:19:13.103625Z","submitted_at":"2024-07-18T17:59:01Z","title":"NNsight and NDIF: Democratizing Access to Open-Weight Foundation Model Internals","version":4},"cited_work":{"arxiv_id":"2407.14561","doi":"10.48550/arxiv.2407.14561","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.14561","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nnsight and ndif: Democratizing access to foundation model internals","venue":"arXiv (Cornell University)","work_id":"3f3a2203-288f-45fe-93e2-436a542d0424","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2407.14561","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:86faad058964e08a724d710a3727e0559cfce47428aa183da545eedc593e2162","observation_id":"e2ad1504-d358-41d6-ad21-1cae28825b56","resolution":{"observed_at":"2026-05-11T10:16:07.456977Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-03T19:08:40.601652+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T19:08:40.601652+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Predictability and surprise in large generative models","venue":null,"work_id":"c9d51f51-c94a-4063-bcb3-9d3c24ed7e44","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:eb23aaa663577616a693cef8292345fc98a5ccd73e82b86e002c76726177b205","observation_id":"c3e3afa1-a371-4125-8327-5cc9424ff23e","resolution":{"observed_at":"2026-05-17T20:02:06.056948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transformer feed-forward layers build predictions by promoting concepts in the vocabulary space","venue":null,"work_id":"1f48b122-8e71-489d-8b66-ceca076ee0c8","year":2022},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:71068ae77cc6e2f5e233c1e485aaa28734fc0c2b821c72d10d0ddd88ce4f1523","observation_id":"5f2dfd27-ea50-438f-9d5d-94ec11faa3d5","resolution":{"observed_at":"2026-05-17T20:02:06.045950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17887","last_updated":"2025-03-03T17:02:05Z","snapshot_observed_at":"2026-08-13T00:44:30.470984Z","submitted_at":"2024-03-26T17:20:04Z","title":"The Unreasonable Ineffectiveness of the Deeper Layers","version":2},"cited_work":{"arxiv_id":"2403.17887","doi":"10.48550/arxiv.2403.17887","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17887","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The unreasonable ineffectiveness of the deeper layers.arXiv preprint arXiv:2403.17887","venue":"arXiv (Cornell University)","work_id":"93761c3c-6e0f-41d0-a8ea-17b8d13c3387","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2403.17887","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:26993e1afbf899adaf13bc0139f055a63c7cb934bfc3c5bed8457036bb75aaf7","observation_id":"d33c3ed4-d16b-4b5c-9096-fdb75cea654a","resolution":{"observed_at":"2026-05-11T10:16:07.468222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.18871","doi":"10.48550/arxiv.2510.18871","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"How do llms use their depth?arXiv preprint arXiv:2510.18871","venue":"arXiv (Cornell University)","work_id":"776ea885-519f-4790-8e33-cac642cc672d","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:1603dd9bb735fbab9d904e11c366e03e2f4e4b1aefa707f23e21e05aa4a72682","observation_id":"1f6eae49-f688-4d33-9b9e-7f8389eb3ac0","resolution":{"observed_at":"2026-05-11T10:16:07.511536Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09476","last_updated":"2024-03-12T07:00:02Z","snapshot_observed_at":"2026-08-10T20:04:32.193409Z","submitted_at":"2023-07-18T17:56:50Z","title":"Overthinking the Truth: Understanding how Language Models Process False Demonstrations","version":3},"cited_work":{"arxiv_id":"2307.09476","doi":"10.48550/arxiv.2307.09476","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.09476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Overthinking the truth: Understanding how language models process false demonstrations","venue":"arXiv (Cornell University)","work_id":"17fcc3f0-2936-4870-bfd6-d55d6dcb949a","year":2023},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2307.09476","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:ffe71d6563a1f56fb21b7861f63856ca536418f1db1f1d781a34534443c9af83","observation_id":"96dfb280-c5f2-4c4a-9f11-91cdeff79d60","resolution":{"observed_at":"2026-05-11T10:16:07.478892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.14064","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"What affects the effective depth of large language models?","venue":null,"work_id":"1ca67695-3c34-4c56-936e-2ade944d06a0","year":2025},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:027af54f480caeb03c861f3bd3b855f158b980138a13a7ce41d84485ceaee08e","observation_id":"76acdfda-3577-4f59-b503-954295a0467d","resolution":{"observed_at":"2026-05-11T10:16:07.586348Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19384","last_updated":"2025-06-16T10:21:00Z","snapshot_observed_at":"2026-08-12T23:33:24.854883Z","submitted_at":"2024-06-27T17:57:03Z","title":"The Remarkable Robustness of LLMs: Stages of Inference?","version":3},"cited_work":{"arxiv_id":"2406.19384","doi":"10.48550/arxiv.2406.19384","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.19384","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2406.19384 , year=","venue":"arXiv (Cornell University)","work_id":"0693ff02-dd9d-4b9b-b17d-a09c0c5f9c52","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2406.19384","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:47cc417809a5d74f7e29e33e2e737f8eb85dfdf26a34aa43e65a522594d1914b","observation_id":"3d28ba6f-b72b-4a56-9343-362ffd474b46","resolution":{"observed_at":"2026-05-11T10:16:07.389361Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Racing thoughts: Explaining contextualization errors in large language models","venue":null,"work_id":"50a95083-842c-4bdb-ac89-82447ac8fede","year":2025},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:f63077c82eb51af08d008717d44983349bb38f3517c984d5f704ec044fc0b32a","observation_id":"c1a9323e-886c-45f2-87a7-87a31a45dca7","resolution":{"observed_at":"2026-05-17T20:02:06.049581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19521","last_updated":"2024-05-24T15:06:45Z","snapshot_observed_at":"2026-08-13T00:42:54.257795Z","submitted_at":"2024-03-28T15:54:59Z","title":"Interpreting Key Mechanisms of Factual Recall in Transformer-Based Language Models","version":4},"cited_work":{"arxiv_id":"2403.19521","doi":"10.48550/arxiv.2403.19521","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.19521","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpreting key mechanisms of factual recall in transformer-based language models","venue":"arXiv (Cornell University)","work_id":"74dcb00e-18a4-41d4-9ac5-fc593254c4db","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2403.19521","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:5f6340bfa6a5ec552d35b2500fe7a4bdd759124e98e82cc811594b53d00bc5e2","observation_id":"7d15240a-58ef-4f0b-9bfc-44abbfcdd8b1","resolution":{"observed_at":"2026-05-11T10:16:07.332182Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07923","last_updated":"2024-04-11T18:03:53Z","snapshot_observed_at":"2026-08-13T05:51:44.514933Z","submitted_at":"2023-10-11T22:35:18Z","title":"The Expressive Power of Transformers with Chain of Thought","version":5},"cited_work":{"arxiv_id":"2310.07923","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.07923","snapshot_observed_at":"2026-07-10T22:47:37.037905Z","title":"and Sabharwal, A","venue":"cs.LG","work_id":"63096903-9f44-418b-b53b-f7debf28d7a9","year":2023},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2310.07923","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:28b81bf8231662750d89f4e0d3cfa187e61066d13380d780c72a2282b61a451a","observation_id":"b412cea5-9305-4354-9159-b7ed227b7048","resolution":{"observed_at":"2026-05-11T10:16:07.576268Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.03961","doi":"10.48550/arxiv.2503.03961","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A little depth goes a long way: The expressive power of log-depth transformers.CoRR, abs/2503.03961","venue":"ArXiv.org","work_id":"3cb572e3-fee9-4fc2-956f-848c1caa91e2","year":2025},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:4a3173401e02647928f1ac2f791f699701a7e56e95a6399dc03f5c233fa0e9d9","observation_id":"2591b39e-bd0a-46f2-b869-1e2c176a639f","resolution":{"observed_at":"2026-05-11T10:16:07.407695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Language models implement simple word2vec- style vector arithmetic","venue":null,"work_id":"d966c7df-e0b8-47d8-b76d-c5e614952cfd","year":2024},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:aed7d36c82e7aa2dee1abc7d762e8843f2b6f41a60ba93842374d57b5ca04d42","observation_id":"13da4258-01d0-41d1-b289-ecbca14f3bb4","resolution":{"observed_at":"2026-05-17T20:02:06.039148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.06477","doi":"10.48550/arxiv.2510.06477","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2510.06477 , year=","venue":"ArXiv.org","work_id":"96bbe6e4-8e47-4301-a6e2-72c0606a8004","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:5d73d8fb9c9c8778139c04b85d3fdff84099c1333dac4b268c30213186cd46e0","observation_id":"7a613b11-0755-4a0c-b9ff-f1f06654101d","resolution":{"observed_at":"2026-05-11T10:16:07.553758Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Understanding transformer reasoning capabilities via graph algorithms.Advances in Neural Information Processing Systems, 37:78320–78370","venue":null,"work_id":"049bc4b5-d91a-43b0-884c-a39ddb54f35a","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:9d275d092e3553aacacb7f050c55ea766c5e3bc9eb4860803b4a34a5b7bb8d5a","observation_id":"b985359c-6e21-49ca-b091-11eeda09e607","resolution":{"observed_at":"2026-05-17T20:02:06.042779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17416","last_updated":"2025-02-24T18:49:05Z","snapshot_observed_at":"2026-08-12T17:52:51.001662Z","submitted_at":"2025-02-24T18:49:05Z","title":"Reasoning with Latent Thoughts: On the Power of Looped Transformers","version":1},"cited_work":{"arxiv_id":"2502.17416","doi":"10.48550/arxiv.2502.17416","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.17416","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2502.17416 (2025)","venue":"ArXiv.org","work_id":"76f16ec7-dba4-4c60-be1b-d9036142c3b7","year":2025},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2502.17416","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:d1a1b6420061a2c94b0b233ce9b5a40a86df290d4cc83562fc296dc310fa4f52","observation_id":"c46d9110-389b-429b-b60f-141fa562ffd6","resolution":{"observed_at":"2026-05-11T10:16:07.361356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-01T13:38:11.849962+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T13:38:11.849962+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1908.06177","last_updated":"2019-09-04T00:14:56Z","snapshot_observed_at":"2026-08-12T13:33:22.070759Z","submitted_at":"2019-08-16T21:12:15Z","title":"CLUTRR: A Diagnostic Benchmark for Inductive Reasoning from Text","version":2},"cited_work":{"arxiv_id":"1908.06177","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1908.06177","snapshot_observed_at":"2026-07-02T20:57:23.984225Z","title":"Clutrr: A diagnostic benchmark for inductive reasoning from text","venue":null,"work_id":"68256d87-be6d-4efa-9bcc-daf3b840de6e","year":1908},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/1908.06177","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:a403f69511cec97cf3d225ebdafd2f8df2ab35fc4260e40eed43df819321794c","observation_id":"265b363a-d799-4ec7-a651-a82870d5be20","resolution":{"observed_at":"2026-05-11T10:16:07.545195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.07682","last_updated":"2022-10-26T05:06:24Z","snapshot_observed_at":"2026-08-02T15:56:35.249569Z","submitted_at":"2022-06-15T17:32:01Z","title":"Emergent Abilities of Large Language Models","version":2},"cited_work":{"arxiv_id":"2206.07682","doi":"10.48550/arxiv.2206.07682","metadata_source":"pith","pith_arxiv_id":"2206.07682","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emergent Abilities of Large Language Models","venue":"cs.CL","work_id":"6ea3375b-837c-4640-a175-be7525aa3c6d","year":2022},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2206.07682","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:eae6f820b7c2a6f518e859964c08aeacc6f33328e7a43526d70175fe67750d29","observation_id":"f15ec79b-23bf-4366-aa9c-963d2699961d","resolution":{"observed_at":"2026-05-11T10:16:07.521598Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-07T00:09:07.34152+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-07T00:09:07.34152+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transformers: State-of-the-art natural language processing","venue":null,"work_id":"49d64c49-24b3-4eb8-af8d-e33ee61ee171","year":2020},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:8b905e61e7cbd8685f1d4ea6348a0b5b66f928aa97db78d13f9cb9e34506f2af","observation_id":"3337a277-a7f9-4a89-b829-3ab35601ceeb","resolution":{"observed_at":"2026-05-17T20:02:06.067600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20896","last_updated":"2025-05-30T18:08:50Z","snapshot_observed_at":"2026-08-09T11:08:42.370443Z","submitted_at":"2025-05-27T08:39:20Z","title":"How Do Transformers Learn Variable Binding in Symbolic Programs?","version":2},"cited_work":{"arxiv_id":"2505.20896","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.20896","snapshot_observed_at":"2026-07-04T13:09:51.104464Z","title":"How do transformers learn variable binding in symbolic programs?arXiv preprint arXiv:2505.20896","venue":null,"work_id":"5b9d904c-68c5-41e1-900f-948064f66270","year":2025},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2505.20896","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:d3f1dc8e31a97b6d0abdc5c11cb086fa7fe7318482c968957c9eb6c1194be876","observation_id":"faa8ded1-7880-4a76-9615-01489e70ff03","resolution":{"observed_at":"2026-05-11T10:16:07.537276Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16042","last_updated":"2024-01-17T04:07:06Z","snapshot_observed_at":"2026-08-08T04:58:14.317234Z","submitted_at":"2023-09-27T21:53:56Z","title":"Towards Best Practices of Activation Patching in Language Models: Metrics and Methods","version":2},"cited_work":{"arxiv_id":"2309.16042","doi":"10.48550/arxiv.2309.16042","metadata_source":"pith","pith_arxiv_id":"2309.16042","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards Best Practices of Activation Patching in Language Models: Metrics and Methods","venue":"cs.LG","work_id":"eead806e-b03c-49ee-91d8-a5f737969604","year":2023},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"cited_paper":"/paper/2309.16042","citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:903fe42f4f3275cfb59d7965f3175d2474f64031bc5d79d37ebe8301c3bca1c0","observation_id":"60d57098-151a-42fb-9c05-7ac244a1f871","resolution":{"observed_at":"2026-05-17T11:56:11.458226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"31ced5ac-a111-4f9b-b39d-07b466fe59e6","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:2d1ddfff86388022df7e2f325a21258a29ff74dee3bc3c0bba49c6c6265597d1","observation_id":"ef8ae413-68fb-45d2-9daf-e5e2e9585bc6","resolution":{"observed_at":"2026-05-17T20:02:06.060407Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Table 1: All pretrained models used in this study, with HuggingFace identifiers, parameter counts, and number of transformer layers","venue":null,"work_id":"d59ec9d7-238d-4353-ba0e-d0b60744e2c1","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:b4d935051d5af6f503ad35ff923794a4e99d21562f23e09bcbe3994224f7e946","observation_id":"8695217e-518f-49da-8c7d-4847ea6f42a2","resolution":{"observed_at":"2026-05-17T20:02:06.064022Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The first row is identical to Fig","venue":null,"work_id":"59538c43-2eb3-4b9a-b02a-a7ca795bda4c","year":2026},"citing_paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T15:34:02.796850Z"},"links":{"citing_paper":"/paper/2604.12426"},"observation_digest":"sha256:f0ce6613e3aa247a138b6802e3f567821da97a016e5addbc7fff989f817a16bd","observation_id":"d61d8620-5b7b-4231-914b-b3b0c9439e6f","resolution":{"observed_at":"2026-05-17T20:02:06.035798Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.12426","last_updated":"2026-04-14T08:16:49Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-10T23:19:59.557920Z","submitted_at":"2026-04-14T08:16:49Z","title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":1,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":1,"verified_exact":16,"verified_fuzzy":8},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 0 inbound Pith citation observations for arXiv:2604.12426."}