{"as_of":"2026-08-05T18:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:18be009926dde154d3b7910e3be4e27ca2e915360bccbb4094b4b2c94dd8bb58","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":38,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T02:32:05.153277Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2406.11717","last_updated":"2024-10-30T18:57:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T16:36:12Z","title":"Refusal in Language Models Is Mediated by a Single Direction","version":3},"reference_index":173,"source":"arxiv_source","source_observed_at":"2026-05-13T10:47:55.934081Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2406.11717"},"observation_digest":"sha256:e045b7fbfc68e31959311c71809aa5c2570b76d97451e755cdfbc236cbbf3747","observation_id":"23c0e894-68e5-428f-b2d2-5cecdb0b491d","resolution":{"observed_at":"2026-05-13T10:47:56.080794Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2407.12772","last_updated":"2025-05-05T04:48:45Z","snapshot_observed_at":"2026-07-06T18:47:56.109836Z","submitted_at":"2024-07-17T17:51:53Z","title":"LMMs-Eval: Reality Check on the Evaluation of Large Multimodal Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T05:19:22.423762Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2407.12772"},"observation_digest":"sha256:b4e9037fc7a39589d9c4bfe9533bc3ead89057356b50f89006a2ce18a0f6b8e4","observation_id":"a858ae38-1d42-40f8-a4b0-022b15c8c58e","resolution":{"observed_at":"2026-05-17T05:19:22.473153Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2502.08943","last_updated":"2026-05-10T20:30:05Z","snapshot_observed_at":"2026-07-06T20:35:44.339599Z","submitted_at":"2025-02-13T03:43:33Z","title":"Beyond the Singular: Revealing the Value of Multiple Generations in Benchmark Evaluation","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-23T03:42:30.220989Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2502.08943"},"observation_digest":"sha256:f3c975bf784aedd85968c6affbd8c754aac223e48c70cf856ddfeedd2272f412","observation_id":"3cdd8a53-8bd8-4ec4-bb25-68b8e5613c90","resolution":{"observed_at":"2026-05-23T03:45:21.881227Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2503.02574","last_updated":"2026-05-18T17:54:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-04T12:55:07Z","title":"LLM-Safety Evaluations Lack Robustness","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-23T01:26:45.402983Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2503.02574"},"observation_digest":"sha256:0c143e6806e6b12f19c6c66ec978050ad132706ffaae80d2c1206ba08be7fe91","observation_id":"96a60ddf-57d1-4ac5-8a55-b1319e00fd58","resolution":{"observed_at":"2026-05-23T01:27:21.237513Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2506.02153","last_updated":"2025-09-15T22:15:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-02T18:35:16Z","title":"Small Language Models are the Future of Agentic AI","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-16T11:55:50.897500Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2506.02153"},"observation_digest":"sha256:a10fc8fa25a7c9e3a5ebb612cc3db36c9d0d4be1d60cbcee094d94658012fdef","observation_id":"1cde4c61-88c4-4323-91dd-789a8c441a12","resolution":{"observed_at":"2026-05-16T11:55:50.972225Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2507.02850","last_updated":"2026-04-20T16:20:19Z","snapshot_observed_at":"2026-07-06T21:51:52.256162Z","submitted_at":"2025-07-03T17:55:40Z","title":"LLM Hypnosis: Exploiting User Feedback for Unauthorized Knowledge Injection to All Users","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-19T05:58:17.452837Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2507.02850"},"observation_digest":"sha256:a31972d6eba08261d7c4c7c5d1fe999b702af635b1d651dfe5367361f876b48f","observation_id":"28acc0ee-d2c0-4e64-a7ce-0c9b8f526a57","resolution":{"observed_at":"2026-05-19T06:02:07.842309Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2509.24186","last_updated":"2026-04-06T07:24:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-29T02:06:13Z","title":"Measuring Competency, Not Performance: Item-Aware Evaluation Across Medical Benchmarks","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-18T13:16:19.744864Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2509.24186"},"observation_digest":"sha256:e6a84d9ecaa6255b46fdefc31cbd78b7c6669528f48cd540822dc55fdeea7308","observation_id":"3687ec11-4f3e-44fc-830e-ae480d056225","resolution":{"observed_at":"2026-05-18T13:16:24.095399Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2510.04309","last_updated":"2026-05-16T04:44:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-05T18:05:28Z","title":"Activation Steering with a Feedback Controller","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-21T21:50:05.186376Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2510.04309"},"observation_digest":"sha256:55d2e072dac2b9e60873f0c4b8463f7e3e2f97664cf0c28f44cdc246a3ada09e","observation_id":"48a3726e-5ed3-4e65-b1cb-d09fa4b169ce","resolution":{"observed_at":"2026-05-21T21:50:41.340553Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-03T09:39:13.877212Z","title":"Qwen Team","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2601.13300","last_updated":"2026-06-25T15:25:41Z","snapshot_observed_at":"2026-08-03T09:39:12.002221Z","submitted_at":"2026-01-19T18:56:08Z","title":"OI-Bench: An Option Injection Benchmark for Evaluating LLM Susceptibility to Directive Interference","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T09:39:13.877212Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2601.13300"},"observation_digest":"sha256:90bc77d2cc385fd5a7eed71d99a4d94c844161451b5bf2fda42d6ef3675bec09","observation_id":"715633ba-fe23-42a6-88b9-f2bbad4fb6ff","resolution":{"observed_at":"2026-08-03T09:39:13.877212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2601.20251","last_updated":"2026-05-08T20:35:11Z","snapshot_observed_at":"2026-07-06T22:43:17.026472Z","submitted_at":"2026-01-28T04:59:20Z","title":"Efficient Evaluation of LLM Performance with Statistical Guarantees","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T10:58:40.958435Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2601.20251"},"observation_digest":"sha256:5016321662394d5b1967b07c540c21c93c3d71b51882dd3f677edde389d9c38a","observation_id":"f8c3bb8d-f922-48d4-b2f3-d4fc817460f9","resolution":{"observed_at":"2026-05-16T11:00:51.726570Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-03T06:01:58.615389Z","title":"M., Weber, L., Choshen, L., Sun, Y ., Xu, G., and Yurochkin, M","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.00710","last_updated":"2026-06-22T09:09:57Z","snapshot_observed_at":"2026-08-04T21:27:22.990462Z","submitted_at":"2026-01-31T13:11:39Z","title":"Learning More from Less: Unlocking Internal Representations for Benchmark Compression","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T06:01:58.615389Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2602.00710"},"observation_digest":"sha256:ec30cf7dd0893401872c80ed6e651f170dab7f014f2054287711187bb257e987","observation_id":"d118dd08-a9ed-402e-bd87-6793e5ecf684","resolution":{"observed_at":"2026-08-03T06:01:58.615389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2604.11328","last_updated":"2026-04-13T11:31:04Z","snapshot_observed_at":"2026-08-02T16:37:23.760294Z","submitted_at":"2026-04-13T11:31:04Z","title":"Select Smarter, Not More: Prompt-Aware Evaluation Scheduling with Submodular Guarantees","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T15:47:45.677717Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2604.11328"},"observation_digest":"sha256:eac0694bceb538ba0429e27ececb4fcb4314547964582db52e129ca1e9590a05","observation_id":"4f43890e-7482-4390-8fa6-89024502c78e","resolution":{"observed_at":"2026-05-11T09:50:59.586582Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2604.18543","last_updated":"2026-06-10T02:43:26Z","snapshot_observed_at":"2026-07-06T23:05:21.974248Z","submitted_at":"2026-04-20T17:36:49Z","title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","version":3},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-05-10T04:25:27.381142Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2604.18543"},"observation_digest":"sha256:be7d7395452d3b465cb9d9122e3b35c3f3f562d83e5b8e266ddce277090f9760","observation_id":"65884e4b-0b9f-4958-b4d9-11e5dc1f3602","resolution":{"observed_at":"2026-05-11T11:56:31.186666Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2604.23099","last_updated":"2026-06-01T21:43:03Z","snapshot_observed_at":"2026-07-06T23:09:23.591544Z","submitted_at":"2026-04-25T01:33:57Z","title":"ProEval: Proactive Failure Discovery and Efficient Performance Estimation for Generative AI Evaluation","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-08T08:21:55.648930Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2604.23099"},"observation_digest":"sha256:5acde480aab419a7a116813caf2c5b0784f62a5cb133ff96606a61836dbb83f7","observation_id":"c4b1c7dd-001e-4c8a-9fb3-f6fe57fc1339","resolution":{"observed_at":"2026-05-11T20:41:10.344342Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2605.00022","last_updated":"2026-04-20T00:57:31Z","snapshot_observed_at":"2026-07-06T23:13:33.799847Z","submitted_at":"2026-04-20T00:57:31Z","title":"Putting HUMANS first: Efficient LAM Evaluation with Human Preference Alignment","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T05:50:05.920842Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2605.00022"},"observation_digest":"sha256:0405306ab97ef848043e4f93516c32b6b769b7c9688e44dbe69ade1248e62f25","observation_id":"08b2782c-5c22-4e3e-b7b6-a4efd02b3113","resolution":{"observed_at":"2026-05-10T05:51:09.782572Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2605.01167","last_updated":"2026-05-01T23:52:54Z","snapshot_observed_at":"2026-07-06T23:14:29.125042Z","submitted_at":"2026-05-01T23:52:54Z","title":"Minimizing Collateral Damage in Activation Steering","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-09T18:58:30.056670Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2605.01167"},"observation_digest":"sha256:a9b3cd769d4c27133e2f391953f31851ae32b08a7b9380bc6ed4c92af1b52ddd","observation_id":"542617dd-1eff-4d5a-8380-8b8024bf898e","resolution":{"observed_at":"2026-05-11T15:56:48.964867Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2605.06213","last_updated":"2026-05-26T15:14:12Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T13:15:31Z","title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-08T10:13:35.777910Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2605.06213"},"observation_digest":"sha256:f8f43f132161dfd5530f7476ea0c9b44183fd85ba55eebad7976e0da891b9cae","observation_id":"04920c39-ade0-40b9-91e2-44758f2239be","resolution":{"observed_at":"2026-05-11T20:06:13.759114Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2605.26409","last_updated":"2026-05-26T00:36:42Z","snapshot_observed_at":"2026-08-02T10:55:21.817292Z","submitted_at":"2026-05-26T00:36:42Z","title":"Jailbreak susceptibility prediction and mitigation via the behavioral geometry of models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-29T17:43:47.849960Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2605.26409"},"observation_digest":"sha256:e9ef60a25284db04f1611364b023e22fbb5371c5695125dc9a82f8b47e858e1a","observation_id":"1aa01361-75d2-4f56-a12e-30cdb76ff703","resolution":{"observed_at":"2026-06-29T17:53:47.703641Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2605.30916","last_updated":"2026-05-29T07:01:38Z","snapshot_observed_at":"2026-07-06T23:40:07.833692Z","submitted_at":"2026-05-29T07:01:38Z","title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T23:49:57.580051Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2605.30916"},"observation_digest":"sha256:9b4b4887f1956e02da3c033e910aaadd1854aa240c7eb6a2a039927d1d5c16e2","observation_id":"1abf399e-a176-4f39-b27d-8c6d06d05f65","resolution":{"observed_at":"2026-06-28T23:52:49.484995Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.00494","last_updated":"2026-06-02T02:20:42Z","snapshot_observed_at":"2026-07-06T23:41:10.825760Z","submitted_at":"2026-05-30T02:54:40Z","title":"ProjQ: Project-and-Quantize for Adapter-Aware LLM Compression","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-06-28T19:10:58.845006Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.00494"},"observation_digest":"sha256:a33b58a1e8dd4342d3c287ab41c856c1464afc9fc245ef7ed28b3c368516f700","observation_id":"d25b9345-7f1f-403f-8cc2-6cdf67260e69","resolution":{"observed_at":"2026-06-28T19:12:34.475288Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.01400","last_updated":"2026-05-31T18:45:12Z","snapshot_observed_at":"2026-07-06T23:41:58.121923Z","submitted_at":"2026-05-31T18:45:12Z","title":"Consistent and Distinctive: LLM Benchmark Efficiency via Maximum Independent Set Prompt Selection on Similarity Graphs","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-28T16:56:29.536912Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.01400"},"observation_digest":"sha256:2ebed0860f577613c411806db594f4c72f4fdfa6b51aa6e03a12cc348d543734","observation_id":"1544c1b7-2e17-4048-9fbe-8589d2ae9473","resolution":{"observed_at":"2026-06-28T17:02:24.486335Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.05029","last_updated":"2026-06-03T15:57:42Z","snapshot_observed_at":"2026-08-01T16:51:47.025025Z","submitted_at":"2026-06-03T15:57:42Z","title":"Validity Threats for Foundation Model Research","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-28T06:52:41.653304Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.05029"},"observation_digest":"sha256:56a52362bf05dc73f34401046b279bdc1dfc17f81092e760b25820ce83a69cf4","observation_id":"e736fccb-b1dc-4c3d-b76b-6e31402ef5c6","resolution":{"observed_at":"2026-07-02T07:36:44.859788Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.07616","last_updated":"2026-05-29T20:55:15Z","snapshot_observed_at":"2026-07-06T23:47:15.041928Z","submitted_at":"2026-05-29T20:55:15Z","title":"Item Response Scaling Laws: A Measurement Theory Approach for Efficient and Generalizable Neural Scaling Estimation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T22:48:20.193699Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.07616"},"observation_digest":"sha256:1d3ebda2e93862369fd971a79a15db9d3a3e49f90f5b731d84383c85f56738bb","observation_id":"0a3aa7fa-d48e-4be2-b8ba-7dae1bbc6520","resolution":{"observed_at":"2026-06-28T22:52:45.592125Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.12344","last_updated":"2026-06-10T17:16:23Z","snapshot_observed_at":"2026-08-02T03:07:28.486057Z","submitted_at":"2026-06-10T17:16:23Z","title":"Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T10:39:51.630229Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.12344"},"observation_digest":"sha256:62023fa7e912a67e275b3052166132647d97d9e3afd27c9bf163a6c60fc7ec76","observation_id":"5c73fc89-5080-46de-8b55-9bc5acde3551","resolution":{"observed_at":"2026-07-03T08:57:47.635188Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.14516","last_updated":"2026-06-12T14:47:37Z","snapshot_observed_at":"2026-07-06T23:52:28.557859Z","submitted_at":"2026-06-12T14:47:37Z","title":"Every Eval Ever: A Unifying Schema and Community Repository for AI Evaluation Results","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-27T04:45:20.445703Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.14516"},"observation_digest":"sha256:fd2a0c33e36716ff190408aace5f66e2e7d3b01a2023b5db1e87d89af16c0bae","observation_id":"67372468-6ce8-4916-909a-35b35b35fc27","resolution":{"observed_at":"2026-07-03T16:58:43.293701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.22437","last_updated":"2026-06-25T12:10:03Z","snapshot_observed_at":"2026-08-03T04:49:13.185870Z","submitted_at":"2026-06-21T10:57:43Z","title":"MMGist: A Comprehensive Multimodal Benchmark for 2027","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T11:05:14.573386Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.22437"},"observation_digest":"sha256:0e6ddf5f2d136ccdae749511d965210cbf65233d8dfafe09607f150b92ab99dc","observation_id":"5324da0f-33cc-4946-a04a-de14cd5cb161","resolution":{"observed_at":"2026-07-04T08:49:41.646194Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2606.26079","last_updated":"2026-06-24T17:53:26Z","snapshot_observed_at":"2026-08-01T23:24:51.854552Z","submitted_at":"2026-06-24T17:53:26Z","title":"Same Evidence, Different Answer: Auditing Order Sensitivity in Multimodal Large Language Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-25T19:58:23.594907Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2606.26079"},"observation_digest":"sha256:e35d8c6a1b4a084f40d3a7af845c61fd9685cfa27a2757c97b7d7d2d28e9d371","observation_id":"384563cc-7ff5-4176-a42f-ded7d0652a5c","resolution":{"observed_at":"2026-07-04T20:40:07.541242Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2607.01152","last_updated":"2026-07-02T01:52:53Z","snapshot_observed_at":"2026-08-02T10:02:20.565193Z","submitted_at":"2026-07-01T16:31:11Z","title":"AGC-Bench: Measuring Artificial General Creativity","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-07-02T12:33:53.029578Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.01152"},"observation_digest":"sha256:1518defade9d10bfb173429de41b4527ba8499a50562f2a10ca7998858d07598","observation_id":"2aa155b7-9468-4bbb-b8c3-46635b803cb7","resolution":{"observed_at":"2026-07-02T12:36:56.060287Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2607.01152","last_updated":"2026-07-02T01:52:53Z","snapshot_observed_at":"2026-08-02T10:02:20.565193Z","submitted_at":"2026-07-01T16:31:11Z","title":"AGC-Bench: Measuring Artificial General Creativity","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-07-03T21:25:14.030920Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.01152"},"observation_digest":"sha256:ed2c1436c962783b152d1f5205a1229f92cfc1159f370c65f893c045c6097e83","observation_id":"54b9154a-cbb3-497a-b809-0d1e1887101a","resolution":{"observed_at":"2026-07-03T21:28:58.086500Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2607.01579","last_updated":"2026-07-02T01:23:02Z","snapshot_observed_at":"2026-08-02T15:54:32.616830Z","submitted_at":"2026-07-02T01:23:02Z","title":"OmniPilot: An Uncertainty-Aware LLM Inference Advisor for Heterogeneous GPU Clusters","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-03T06:30:30.308713Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.01579"},"observation_digest":"sha256:981940ec6bc2d07675aa13ae620c09797a72c68498799764b85e3d3c75ce2571","observation_id":"ac233903-70d6-4f06-93c1-39640b514312","resolution":{"observed_at":"2026-07-03T06:37:42.188290Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2607.02007","last_updated":"2026-07-02T10:43:06Z","snapshot_observed_at":"2026-08-03T01:38:17.944886Z","submitted_at":"2026-07-02T10:43:06Z","title":"EduArt: An educational-level benchmark for evaluating art history knowledge in large language models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-07-03T14:48:06.514693Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.02007"},"observation_digest":"sha256:2d32a2c0f8b56295d1cd2b3e050ed00abcb846ec37411a6cd4b7ddbbceb478fb","observation_id":"29a58174-185b-410b-bdb0-3459e2be244f","resolution":{"observed_at":"2026-07-03T14:48:32.254103Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2607.07918","last_updated":"2026-07-08T21:03:27Z","snapshot_observed_at":"2026-08-05T10:24:34.449076Z","submitted_at":"2026-07-08T21:03:27Z","title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-07-10T15:26:23.290009Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.07918"},"observation_digest":"sha256:7c03a23fc42f433b8e5809c339a0e2d5bd2c596783444dd93c8252fd9038b4ff","observation_id":"4503ec75-b14e-4b7c-b05f-561353a21df0","resolution":{"observed_at":"2026-07-10T15:27:20.086993Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":"2402.14992","doi":"10.48550/arxiv.2402.14992","metadata_source":"pith","pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"David Rein, Betty Li Hou, Asa Cooper Stickland, Jack- son Petty, Richard Yuanzhe Pang, Julien Dirani, Ju- lian Michael, and Samuel R Bowman","venue":"cs.CL","work_id":"1aaa40b4-1321-4f79-85a8-e296eaa2b732","year":2024},"citing_paper":{"arxiv_id":"2607.08347","last_updated":"2026-07-09T10:52:24Z","snapshot_observed_at":"2026-07-12T23:18:27.969196Z","submitted_at":"2026-07-09T10:52:24Z","title":"Prediction-Powered Active Testing","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-07-10T08:59:10.780536Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.08347"},"observation_digest":"sha256:055923e566b49eb390fc14fb0dad4b316d3d5c3fa89f5bc3a498fc233629bec0","observation_id":"f5e8326e-8f62-4e2e-a43a-a5d02eaba7a3","resolution":{"observed_at":"2026-07-10T09:06:59.145174Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-07-14T16:38:50.302371Z","title":"Valentina Pyatkin, Saumya Malik, Victoria Graf, Hamish Ivison, Shengyi Huang, Pradeep Dasigi, Nathan Lambert, and Hanna Hajishirzi","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.09739","last_updated":"2026-07-02T18:37:18Z","snapshot_observed_at":"2026-07-31T18:10:50.881034Z","submitted_at":"2026-07-02T18:37:18Z","title":"Coresets Before Score Sets: Evaluation-Unsupervised Prompt Subset Selection for LLM Benchmarks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-14T16:38:50.302371Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.09739"},"observation_digest":"sha256:47e0cb2c5104de9def0e514f07bd54f43a054c3b0c3a8d1bdaabc33350c3fe81","observation_id":"d1a20664-1446-4ebe-925c-e8ca3f0760c1","resolution":{"observed_at":"2026-07-14T16:38:50.302371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-01T14:50:59.373155Z","title":"Ruchir Puri, David S","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.18642","last_updated":"2026-07-21T02:23:22Z","snapshot_observed_at":"2026-08-05T00:33:06.675723Z","submitted_at":"2026-07-21T02:23:22Z","title":"Spaghetti Architect: A Contamination-Resistant, By-Construction-Labelled, Multi-Language Code Dataset Generator","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T14:50:59.373155Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.18642"},"observation_digest":"sha256:96451e2694d16cc5f61c857b986b3031c0244d995847863353e5f049d97916d4","observation_id":"1b8d61ba-cc8f-4051-9fdf-a147ec13c4bb","resolution":{"observed_at":"2026-08-01T14:50:59.373155Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-07-31T16:17:41.484228Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24889","last_updated":"2026-07-27T13:03:19Z","snapshot_observed_at":"2026-08-04T11:06:50.375226Z","submitted_at":"2026-07-27T13:03:19Z","title":"GAUGE: Grading Agent-Built Financial Models Without a Golden Answer","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-31T16:17:41.484228Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.24889"},"observation_digest":"sha256:31d5c032426f597d1f811f774f389d579cdbc255384bb40962cd93aa10f904b2","observation_id":"d2146d11-490a-4c0c-acbc-e1f9e6672da9","resolution":{"observed_at":"2026-07-31T16:17:41.484228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-01T07:40:14.904090Z","title":"arXiv preprint arXiv:2402.14992 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.27420","last_updated":"2026-07-29T19:42:26Z","snapshot_observed_at":"2026-08-04T10:55:46.147631Z","submitted_at":"2026-07-29T19:42:26Z","title":"Dimensionality and Measurement Precision in HLE's Multiple-Choice Subset","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-01T07:40:14.904090Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2607.27420"},"observation_digest":"sha256:ff4f2f1d1060c81b72c1255fb0e0a45147fdf8acc9fdabd382e053203bef7d40","observation_id":"04007ee3-631a-47fc-b775-bda4630151e2","resolution":{"observed_at":"2026-08-01T07:40:14.904090Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-04T02:32:05.153277Z","title":"tinybenchmarks: evaluating llms with fewer examples","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00014","last_updated":"2026-06-24T16:14:12Z","snapshot_observed_at":"2026-08-05T17:20:12.754453Z","submitted_at":"2026-06-24T16:14:12Z","title":"CoT-Core: Accelerating LLM Evaluation via CoT-Aware Coreset Selection","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-04T02:32:05.153277Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2608.00014"},"observation_digest":"sha256:3b45d3c00801eefd41d3473b205671d235a917fe7742e6e2a579325ae0ee2e6a","observation_id":"f94f108e-e739-49d0-8054-c2a53b1c4712","resolution":{"observed_at":"2026-08-04T02:32:05.153277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2402.14992/citation-record","integrity":"/paper/2402.14992/integrity","json":"/paper/2402.14992/citation-record.json","paper":"/paper/2402.14992"},"outbound":[],"paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 38 inbound Pith citation observations for arXiv:2402.14992."}