{"as_of":"2026-08-08T03:34:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:aa44bad597153138338f08612c6db91f9a9325a49a0896c8d8ed1e7f5534de75","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:34:27.021795Z","state":"measured"},{"denominator":72,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":72,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":6,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":6,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T06:40:23.654993Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T10:49:46.920258Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"cited_work":{"arxiv_id":"2506.07673","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07673","snapshot_observed_at":"2026-07-22T01:23:30.685296Z","title":"E., and Hardt, M","venue":null,"work_id":"8897e322-7d9b-40d5-9949-3a993e4445e2","year":2025},"citing_paper":{"arxiv_id":"2601.20251","last_updated":"2026-05-08T20:35:11Z","snapshot_observed_at":"2026-07-06T22:43:17.026472Z","submitted_at":"2026-01-28T04:59:20Z","title":"Efficient Evaluation of LLM Performance with Statistical Guarantees","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T10:58:40.958435Z"},"links":{"cited_paper":"/paper/2506.07673","citing_paper":"/paper/2601.20251"},"observation_digest":"sha256:0885c3e9a6ded8a0fa38bf8919e87cd09184d65eb01b23a03d8bc14d972b1f20","observation_id":"b27a9196-c3e0-4eb4-9adc-6acd9af2d40c","resolution":{"observed_at":"2026-07-22T01:23:30.685296Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"cited_work":{"arxiv_id":"2506.07673","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07673","snapshot_observed_at":"2026-07-22T01:23:30.685296Z","title":"E., and Hardt, M","venue":null,"work_id":"8897e322-7d9b-40d5-9949-3a993e4445e2","year":2025},"citing_paper":{"arxiv_id":"2605.05973","last_updated":"2026-05-07T10:18:56Z","snapshot_observed_at":"2026-07-06T23:18:31.432422Z","submitted_at":"2026-05-07T10:18:56Z","title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-08T05:26:41.529866Z"},"links":{"cited_paper":"/paper/2506.07673","citing_paper":"/paper/2605.05973"},"observation_digest":"sha256:a8a8ffc09c952cfdadc85e62b92928ac5cea5f6456d47894ff3ef15b67fae0a6","observation_id":"6992b4b1-fe2e-435b-85da-cacb5393970f","resolution":{"observed_at":"2026-07-22T01:23:30.685296Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"cited_work":{"arxiv_id":"2506.07673","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07673","snapshot_observed_at":"2026-07-22T01:23:30.685296Z","title":"E., and Hardt, M","venue":null,"work_id":"8897e322-7d9b-40d5-9949-3a993e4445e2","year":2025},"citing_paper":{"arxiv_id":"2606.03330","last_updated":"2026-06-02T08:39:50Z","snapshot_observed_at":"2026-08-02T20:44:01.368085Z","submitted_at":"2026-06-02T08:39:50Z","title":"FLIPS: Instance-Fingerprinting for LLMs via Pseudo-random Sequences","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-06-28T11:24:01.547119Z"},"links":{"cited_paper":"/paper/2506.07673","citing_paper":"/paper/2606.03330"},"observation_digest":"sha256:103b94a747e3d1769dccbf6a4e66e4e97291d0556b3c24c6a87920b1c72e0a40","observation_id":"470626fd-38bc-4101-a212-c01eccc4759e","resolution":{"observed_at":"2026-07-22T01:23:30.685296Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"cited_work":{"arxiv_id":"2506.07673","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07673","snapshot_observed_at":"2026-07-22T01:23:30.685296Z","title":"E., and Hardt, M","venue":null,"work_id":"8897e322-7d9b-40d5-9949-3a993e4445e2","year":2025},"citing_paper":{"arxiv_id":"2606.05029","last_updated":"2026-06-03T15:57:42Z","snapshot_observed_at":"2026-08-01T16:51:47.025025Z","submitted_at":"2026-06-03T15:57:42Z","title":"Validity Threats for Foundation Model Research","version":1},"reference_index":110,"source":"pdf_text","source_observed_at":"2026-06-28T06:52:41.653304Z"},"links":{"cited_paper":"/paper/2506.07673","citing_paper":"/paper/2606.05029"},"observation_digest":"sha256:797f221e8e79d80f4eccb573852eb39fbc46bd9ac985275940a355c6796cc477","observation_id":"e131494d-8e10-4c20-bf88-bbc35f486f3d","resolution":{"observed_at":"2026-07-22T01:23:30.685296Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"cited_work":{"arxiv_id":"2506.07673","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.07673","snapshot_observed_at":"2026-07-22T01:23:30.685296Z","title":"E., and Hardt, M","venue":null,"work_id":"8897e322-7d9b-40d5-9949-3a993e4445e2","year":2025},"citing_paper":{"arxiv_id":"2606.24020","last_updated":"2026-06-22T23:54:00Z","snapshot_observed_at":"2026-08-06T15:08:13.351860Z","submitted_at":"2026-06-22T23:54:00Z","title":"You Don't Need to Run Every Eval","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-26T08:23:12.145516Z"},"links":{"cited_paper":"/paper/2506.07673","citing_paper":"/paper/2606.24020"},"observation_digest":"sha256:57b5c7e191aa98b7dfd5c8e3e477cf1e672da048e9812b73ae36de0bbd45c5ff","observation_id":"cf978fc5-69ca-48b4-a646-c55b8a9574af","resolution":{"observed_at":"2026-07-22T01:23:30.685296Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07673","snapshot_observed_at":"2026-08-01T06:40:23.654993Z","title":"How benchmark prediction from fewer data misses the mark.arXiv preprint arXiv:2506.07673, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21839","last_updated":"2026-07-23T22:06:12Z","snapshot_observed_at":"2026-08-01T06:40:17.382657Z","submitted_at":"2026-07-23T22:06:12Z","title":"Certified in Theory, Broken in Practice: Assumption Gaps in Cryptographic Model Certification","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-01T06:40:23.654993Z"},"links":{"cited_paper":"/paper/2506.07673","citing_paper":"/paper/2607.21839"},"observation_digest":"sha256:44f4ff77eeea0a920a50bcad447a02eaaf3733c2bdb59a4354740266a55b61cb","observation_id":"497f56ee-452b-4e85-b947-3c7cecea0230","resolution":{"observed_at":"2026-08-01T06:40:23.654993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.07673/citation-record","integrity":"/paper/2506.07673/integrity","json":"/paper/2506.07673/citation-record.json","paper":"/paper/2506.07673"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.842913Z","title":"Jordan, and Tijana Zrnic","venue":null,"work_id":"24493b14-9fd7-4a29-8a27-ce3883b422d0","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.564352Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:2399d4ed1fef0bb41c5549915294ac25a89ff65ff022cdd98014075eed120fb3","observation_id":"26ce0089-b6d4-4d42-b7b8-b406dacf525f","resolution":{"observed_at":"2026-08-07T05:34:27.846205Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01453","last_updated":"2024-03-26T01:44:52Z","snapshot_observed_at":"2026-07-06T16:42:19.938941Z","submitted_at":"2023-11-02T17:59:04Z","title":"PPI++: Efficient Prediction-Powered Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01453","snapshot_observed_at":"2026-08-07T05:34:26.571156Z","title":"Duchi, and Tijana Zrnic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.571156Z"},"links":{"cited_paper":"/paper/2311.01453","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:45b964e44c3883e4c46423cc564be703a145bb11f3204f67b1107724bf31a8a3","observation_id":"12e946a6-75f0-4505-a47d-68740937a7ac","resolution":{"observed_at":"2026-08-07T05:34:26.571156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1308.3432","last_updated":"2013-08-15T15:19:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2013-08-15T15:19:34Z","title":"Estimating or Propagating Gradients Through Stochastic Neurons for Conditional Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1308.3432","snapshot_observed_at":"2026-08-07T05:34:26.582523Z","title":"Courville","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.582523Z"},"links":{"cited_paper":"/paper/1308.3432","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:4e3ca0ac1821a612ea586343e4b4806006cfe683f2192863de1b3827e5fa81b5","observation_id":"e8998634-c9ec-4efd-a751-da8bce42a030","resolution":{"observed_at":"2026-08-07T05:34:26.582523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:26.595425Z","title":"The fifth PASCAL recognizing textual entailment challenge","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.595425Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:7af1782318e3b281049fa7fe7b1ae2ffaf0320f8bd7c54684d9e3ce8a0d1d860","observation_id":"cb62aaaa-d323-4ab9-b403-cb5c3f267220","resolution":{"observed_at":"2026-08-07T05:34:26.595425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07008","last_updated":"2026-06-01T01:28:08Z","snapshot_observed_at":"2026-08-06T02:41:44.773909Z","submitted_at":"2024-03-09T02:47:11Z","title":"AutoEval Done Right: Using Synthetic Data for Model Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07008","snapshot_observed_at":"2026-08-07T05:34:26.612857Z","title":"Autoeval done right: Using synthetic data for model evaluation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.612857Z"},"links":{"cited_paper":"/paper/2403.07008","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:0433da034477e976be7e663c25a7fb10f48d5e8dd95be7e12dd655d445484154","observation_id":"b456a683-73b7-4e2a-9d30-d37b78bf7543","resolution":{"observed_at":"2026-08-07T05:34:26.612857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:26.629844Z","title":"A singular value thresholding algorithm for matrix completion","venue":null,"work_id":null,"year":1956},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.629844Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:4b3cec14c40d2e34dbfbc6fed5778855abc710eb068b4e6060fed147ee75a5f1","observation_id":"4be3af6d-5e00-46bc-b5ee-b7b40d24cfd3","resolution":{"observed_at":"2026-08-07T05:34:26.629844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.10669","last_updated":"2024-09-26T03:16:52Z","snapshot_observed_at":"2026-07-06T17:31:08.762955Z","submitted_at":"2024-02-16T13:21:06Z","title":"Humans or LLMs as the Judge? A Study on Judgement Biases","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.10669","snapshot_observed_at":"2026-08-07T05:34:26.647558Z","title":"Humans or llms as the judge? a study on judgement biases","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.647558Z"},"links":{"cited_paper":"/paper/2402.10669","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:8a3529fb0b7bae3feb2a06a5e1f357b9ea2e8a126f7bd072ba70ce00b667ff54","observation_id":"0696c877-6cb1-450c-898d-563b3cd76d5b","resolution":{"observed_at":"2026-08-07T05:34:26.647558Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-07T05:34:26.659435Z","title":"Think you have solved question answering? try arc, the ai 2 reasoning challenge","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.659435Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:94b196d3d7d36286ca9723c5a38546a7e5754d81de5b85c0f1be77a0474f987b","observation_id":"4900b0b7-bbd8-482c-b3ee-442914c09442","resolution":{"observed_at":"2026-08-07T05:34:26.659435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T05:34:26.728273Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.728273Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:067782e9a95ad06956b52e7c68fb439844790a9a70af4758a959d872a874525b","observation_id":"4a51dd97-39e8-48af-ba9d-61a528884761","resolution":{"observed_at":"2026-08-07T05:34:26.728273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.819567Z","title":"Computing the testing error without a testing set","venue":null,"work_id":"840c108f-0b78-44f5-9fd1-74087e6bd938","year":2020},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.804792Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:63816c5fdaaa30e5a50bc91c42f325aa5ae1cb4f00fea227b815954da68de12b","observation_id":"aad63229-20d8-44ad-9299-46d04ceaf4c6","resolution":{"observed_at":"2026-08-07T05:34:27.823011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.809653Z","title":"The PASCAL recognising textual entailment challenge","venue":null,"work_id":"a56d1491-9364-48e6-8e3f-30fcc0b2d58c","year":2006},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.845227Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:ff8e0afb8afb5c0e1a63a8a82a4ea918e2f7f86f9f3823148b6ca01ae8cd4b28","observation_id":"b98538e5-9647-4134-87d6-3619b1fe2cbf","resolution":{"observed_at":"2026-08-07T05:34:27.813136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.799190Z","title":"Are labels always necessary for classifier accuracy evaluation? In Proceedings of the IEEE/CVF conference on computer vision and pattern recognition, pages 15069– 15078, 2021","venue":null,"work_id":"3e47649e-7148-4e7a-941e-9f316342fa2b","year":2021},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.848483Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:3a65e026a48c90d9f11f458f29d0391349baaf8d44f0f38d1588ee9349355816","observation_id":"d22a3f24-6499-4e4b-86a0-4f7c3537c121","resolution":{"observed_at":"2026-08-07T05:34:27.803019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.787504Z","title":"Automatically constructing a corpus of sentential paraphrases","venue":null,"work_id":"3f035412-2a72-4851-9b4d-d4301566ccea","year":2005},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.851663Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:dd7d0778fd2a48c9d7dbca96f5c7038ecfd0160701789a0154d83a060fe4bd68","observation_id":"41110b6f-0e2e-4556-bd50-59291e0a0e13","resolution":{"observed_at":"2026-08-07T05:34:27.791643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.775672Z","title":"Limits to scalable evaluation at the frontier: Llm as judge won’t beat twice the data","venue":null,"work_id":"5daef60b-74c1-4c76-ab26-5d086c262423","year":2025},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.854801Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:c246ac779e21bdd4575cd21040d614a83f7125071c32b1e4cd901c46b70af83c","observation_id":"57d1e637-a851-45ce-ac17-aa6b56ab3582","resolution":{"observed_at":"2026-08-07T05:34:27.779352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.765064Z","title":"Facility location: concepts, models, algo- rithms and case studies","venue":null,"work_id":"cb3670a6-7881-4819-81de-9749782a8019","year":2009},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.858336Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:53f2e9391df03f4b6894fe1cf25ad9d1ddb888cb1f8d3adb31bc5370d489bd9b","observation_id":"dce72101-6bdb-446d-9e38-3fdc7196cc92","resolution":{"observed_at":"2026-08-07T05:34:27.768782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.754667Z","title":"Open llm leaderboard v2","venue":null,"work_id":"be072a05-1a27-4118-b58f-b33f0a80546a","year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.861417Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:e2d687ed667185245ad4d56df1b01109a9d4f63896eb887ce3d1154fec14a040","observation_id":"3eebed2e-4b99-46e3-95ad-6207737980b7","resolution":{"observed_at":"2026-08-07T05:34:27.758038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.743906Z","title":"Challenges in evaluating AI systems, 2023","venue":null,"work_id":"9666a697-8e5b-47fa-952d-65686762808c","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.864676Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:4ba764aaf653ef0dd8acd20bf6a89a7c24cdc7259cdacaa0ec8ea540f373eb17","observation_id":"d1bfc293-0b96-49ef-b95b-6e74709768c6","resolution":{"observed_at":"2026-08-07T05:34:27.747484Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.733909Z","title":"The third PASCAL recognizing textual entailment challenge","venue":null,"work_id":"d97ad6be-f31d-448c-92f9-781aeb17435c","year":2007},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.868362Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:9b2afe6565b76b102105a194b7cf82311e80cd52d14b50671a75a35a9c7fbbf5","observation_id":"fef75381-5e12-4c9b-8e76-4763ed271a4e","resolution":{"observed_at":"2026-08-07T05:34:27.737314Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:26.872011Z","title":"An introduction to the augmented inverse propensity weighted estimator","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.872011Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:869c00af8beb7dcd3078963c0e394d420259f5c1722c67c25a4e86fd55ef40de","observation_id":"e92bd0f6-4ca0-4717-9731-1ee70e48c1e3","resolution":{"observed_at":"2026-08-07T05:34:26.872011Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04313","last_updated":"2025-06-12T14:43:36Z","snapshot_observed_at":"2026-07-06T20:32:23.502480Z","submitted_at":"2025-02-06T18:56:01Z","title":"Great Models Think Alike and this Undermines AI Oversight","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04313","snapshot_observed_at":"2026-08-07T05:34:26.874878Z","title":"Chandra, Ponnurangam Kumaraguru, Douwe Kiela, Ameya Prabhu, Matthias Bethge, and Jonas Geiping","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.874878Z"},"links":{"cited_paper":"/paper/2502.04313","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:37f5a4773ff80dc5d9bf0e496f68d4ae20cda987cbefbd5dbc1b335053b21b35","observation_id":"f75e2726-15af-4092-bb23-e4c5e4ba78b6","resolution":{"observed_at":"2026-08-07T05:34:26.874878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15594","last_updated":"2025-10-19T10:32:43Z","snapshot_observed_at":"2026-08-02T10:23:50.881300Z","submitted_at":"2024-11-23T16:03:35Z","title":"A Survey on LLM-as-a-Judge","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15594","snapshot_observed_at":"2026-08-07T05:34:26.877994Z","title":"A survey on llm-as-a-judge","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.877994Z"},"links":{"cited_paper":"/paper/2411.15594","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:353eee5b4ddbe72923bfe18b778338a9a9064e10db0d1e3f7900ae3f7e8b3367","observation_id":"011eccdc-8773-4472-a298-376579d64cd4","resolution":{"observed_at":"2026-08-07T05:34:26.877994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.716126Z","title":"Ho, Christopher Ré, Adam Chilton, Aditya Narayana, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel N","venue":null,"work_id":"96fbbdc1-7d6f-4d5f-a95e-ae0ab6da1a06","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.880977Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:f916637892244d4cf76969f7619f66458362572b1508df2017bece157504f70b","observation_id":"9a4172e1-1add-4cf7-9d9a-ff19aaaa9c42","resolution":{"observed_at":"2026-08-07T05:34:27.719690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02575","last_updated":"2023-08-03T12:47:17Z","snapshot_observed_at":"2026-08-06T01:33:35.207477Z","submitted_at":"2023-08-03T12:47:17Z","title":"Is GPT-4 a reliable rater? Evaluating Consistency in GPT-4 Text Ratings","version":1},"cited_work":{"arxiv_id":"2308.02575","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.02575","snapshot_observed_at":"2026-08-07T05:34:27.478087Z","title":"Is GPT-4 a reliable rater? Evaluating Consistency in GPT-4 Text Ratings","venue":"cs.CL","work_id":"8524a321-58eb-4a2d-982c-d08fcb2b7e7f","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.884061Z"},"links":{"cited_paper":"/paper/2308.02575","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:2487985b21c95f2ee7990a322efceb6a5616c0dcc90acae6eda9136e08a9a2dc","observation_id":"e236eb94-6ade-4c5c-8d7c-372701a630ef","resolution":{"observed_at":"2026-08-07T05:34:27.482516Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.18466","last_updated":"2024-02-02T20:28:27Z","snapshot_observed_at":"2026-08-04T19:50:24.574752Z","submitted_at":"2023-05-29T08:03:28Z","title":"Test-Time Training on Nearest Neighbors for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.18466","snapshot_observed_at":"2026-08-07T05:34:26.887369Z","title":"Test-time training on nearest neighbors for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.887369Z"},"links":{"cited_paper":"/paper/2305.18466","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:acafef9fbee8bc20c90e000b223f393c9cbeb069a7c2bf9a38312c3e614c365f","observation_id":"9d271dc8-53d5-4f79-a88a-b7ed61a6cf7f","resolution":{"observed_at":"2026-08-07T05:34:26.887369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-07T05:34:26.890744Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.890744Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:9a055b97f76b18b4a029326a3b2de26e964c1702ad61c91b82db5c1e2bc7c13f","observation_id":"28058209-5f05-40f2-aee6-3bd86c0af9a6","resolution":{"observed_at":"2026-08-07T05:34:26.890744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-07T05:34:26.894434Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.894434Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:b268240c29a02f39efc840964280faaeb4c5fa6fdca1290d596df4d71d443f9c","observation_id":"e3adc9fc-5581-4631-8821-dd64ac57f168","resolution":{"observed_at":"2026-08-07T05:34:26.894434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1611.01144","last_updated":"2017-08-05T22:45:19Z","snapshot_observed_at":"2026-08-01T18:34:23.156273Z","submitted_at":"2016-11-03T19:48:08Z","title":"Categorical Reparameterization with Gumbel-Softmax","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.01144","snapshot_observed_at":"2026-08-07T05:34:26.897769Z","title":"Categorical reparameterization with gumbel- softmax","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.897769Z"},"links":{"cited_paper":"/paper/1611.01144","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:44f8eec8240432ca2d0c03233f6eeecb8cf0efd4682772c011ec06217d7e1459","observation_id":"f0802203-fea5-4bf8-8d0a-3f8e4ffbd548","resolution":{"observed_at":"2026-08-07T05:34:26.897769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.13081","last_updated":"2020-09-28T05:07:51Z","snapshot_observed_at":"2026-08-06T03:17:52.286711Z","submitted_at":"2020-09-28T05:07:51Z","title":"What Disease does this Patient Have? A Large-scale Open Domain Question Answering Dataset from Medical Exams","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.13081","snapshot_observed_at":"2026-08-07T05:34:26.901004Z","title":"What disease does this patient have? a large-scale open domain question answering dataset from medical exams","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.901004Z"},"links":{"cited_paper":"/paper/2009.13081","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:e1b90afa262e0fe13abe21d5117ea9129268ead2b4fb3feb69d9799760920c3d","observation_id":"87a8b5ae-eb96-4c08-9469-491f0d59427c","resolution":{"observed_at":"2026-08-07T05:34:26.901004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-07T05:34:26.904300Z","title":"Scaling laws for neural language models","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.904300Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:cceeecd295ba82eca12e491bdfcc4dce6fe9b6d9f0262dfc8a9254e3a43fe69c","observation_id":"2f0551a2-b906-4595-bfc7-fbf5e87d54f4","resolution":{"observed_at":"2026-08-07T05:34:26.904300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.706076Z","title":"Active testing: Sample- efficient model evaluation","venue":null,"work_id":"aa123000-7cdb-4567-a1a1-cb6bfe6b83b5","year":2021},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.907237Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:afbee049bc306adce10f0f661c05e5040df463e82da919ba301eb76e72e41434","observation_id":"31b45bc4-f92a-41a9-91b7-4d3ab770d763","resolution":{"observed_at":"2026-08-07T05:34:27.709413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.06881","last_updated":"2022-10-18T18:43:08Z","snapshot_observed_at":"2026-07-06T12:37:39.880827Z","submitted_at":"2022-02-14T17:15:18Z","title":"Active Surrogate Estimators: An Active Learning Approach to Label-Efficient Model Evaluation","version":2},"cited_work":{"arxiv_id":"2202.06881","doi":null,"metadata_source":"pith","pith_arxiv_id":"2202.06881","snapshot_observed_at":"2026-08-07T05:34:27.401653Z","title":"Active Surrogate Estimators: An Active Learning Approach to Label-Efficient Model Evaluation","venue":"cs.LG","work_id":"357e5083-fbda-4634-9983-04cc1851a02e","year":2022},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.909784Z"},"links":{"cited_paper":"/paper/2202.06881","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:b01f0a6006fe0eb1ec3295dde7365f7498d790950b5d9894eaab3d382b883a92","observation_id":"0337a2e4-f01d-4d11-bdf5-60f2c848bf25","resolution":{"observed_at":"2026-08-07T05:34:27.405042Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:26.912846Z","title":"Retrieval- augmented generation for knowledge-intensive nlp tasks","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.912846Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:80b97df176c3ad13227ecc017e98d7128a2959b5ddb2fe32173bd52709c0bd3b","observation_id":"7fc985a9-dc69-49a8-8d29-7754f35819ff","resolution":{"observed_at":"2026-08-07T05:34:26.912846Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05952","last_updated":"2024-10-08T12:08:46Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:08:46Z","title":"Active Evaluation Acquisition for Efficient LLM Benchmarking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05952","snapshot_observed_at":"2026-08-07T05:34:26.915668Z","title":"Active evaluation acquisition for efficient llm benchmarking","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.915668Z"},"links":{"cited_paper":"/paper/2410.05952","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:704743b64aed7265eb2d7fda272abfdd23aba22a3d529b06acce608c4d49cbfd","observation_id":"75fea7a5-cf22-4c32-a251-182db853b17e","resolution":{"observed_at":"2026-08-07T05:34:26.915668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.688983Z","title":"Manning, Christopher R’e, Diana Acosta-Navas, Drew A","venue":null,"work_id":"1766ea4e-a120-444f-9751-9d90499ef3af","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.919274Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:30e3b95d700c0f2d12eba93ee3ae4c64fd32b262f95b2dad0bb15b685a1ff568","observation_id":"423e05f4-9dde-4337-9584-8a968046cef3","resolution":{"observed_at":"2026-08-07T05:34:27.692361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10229","last_updated":"2024-06-14T17:59:54Z","snapshot_observed_at":"2026-07-06T18:31:08.584682Z","submitted_at":"2024-06-14T17:59:54Z","title":"Quantifying Variance in Evaluation Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10229","snapshot_observed_at":"2026-08-07T05:34:26.922228Z","title":"Singh, Rylan Schaeffer, Andrew Poulton, Oluwasanmi Koyejo, Pontus Stenetorp, Sharan Narang, and Dieuwke Hupkes","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.922228Z"},"links":{"cited_paper":"/paper/2406.10229","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:7c96afe2e2a0883dbbe911fb19ce3b933d28a1d2ae3790ddf9e24506224842c2","observation_id":"5705a10f-9f73-4de9-ac11-01fc409cea29","resolution":{"observed_at":"2026-08-07T05:34:26.922228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1905.12580","last_updated":"2019-05-29T16:54:37Z","snapshot_observed_at":"2026-07-06T07:56:30.864880Z","submitted_at":"2019-05-29T16:54:37Z","title":"Model Similarity Mitigates Test Set Overuse","version":1},"cited_work":{"arxiv_id":"1905.12580","doi":null,"metadata_source":"pith","pith_arxiv_id":"1905.12580","snapshot_observed_at":"2026-08-07T05:34:27.363209Z","title":"Model Similarity Mitigates Test Set Overuse","venue":"cs.LG","work_id":"9fdbaba4-356a-42b6-8a23-3c2199f58655","year":2019},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.925314Z"},"links":{"cited_paper":"/paper/1905.12580","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:084514414ca6ee7e6d0ee708e804d73cff1c733b36e1b18fef2b4b23f92ab1b6","observation_id":"c16dfc68-05f7-407b-8512-b586c565db80","resolution":{"observed_at":"2026-08-07T05:34:27.367042Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.678345Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering","venue":null,"work_id":"b20d6a0c-e9fc-4885-b50a-1aed22282f83","year":2018},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.928473Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:83cf1505eeac2157a4dc5972eef984f090f43709def2b280cc24be7b6e394f05","observation_id":"ef43aa4b-2f20-4461-a61c-8771e5acea8b","resolution":{"observed_at":"2026-08-07T05:34:27.682228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04757","last_updated":"2024-01-09T17:34:30Z","snapshot_observed_at":"2026-08-07T05:53:39.752355Z","submitted_at":"2024-01-09T17:34:30Z","title":"How predictable is language model benchmark performance?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04757","snapshot_observed_at":"2026-08-07T05:34:26.931350Z","title":"How predictable is language model benchmark performance? ArXiv, abs/2401.04757, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.931350Z"},"links":{"cited_paper":"/paper/2401.04757","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:1f2cca93b89ef6219b76fee18d6cc7bc4df0bf63a0901ab3d92f4c59b186f128","observation_id":"080a3031-0e61-44d9-a8f4-1d99a863ec85","resolution":{"observed_at":"2026-08-07T05:34:26.931350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14445","last_updated":"2025-06-17T14:34:13Z","snapshot_observed_at":"2026-08-07T18:02:24.171322Z","submitted_at":"2025-02-20T10:52:38Z","title":"PredictaBoard: Benchmarking LLM Score Predictability","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14445","snapshot_observed_at":"2026-08-07T05:34:26.934415Z","title":"Predictaboard: Benchmarking llm score predictability","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.934415Z"},"links":{"cited_paper":"/paper/2502.14445","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:b911f3636523c8ced681da2098dffe8f686f6047056f2981dbead2d370963d8a","observation_id":"bf06c144-35d8-4081-ae7f-9f0e12da39ea","resolution":{"observed_at":"2026-08-07T05:34:26.934415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13076","last_updated":"2024-04-15T16:49:59Z","snapshot_observed_at":"2026-07-06T18:02:53.240463Z","submitted_at":"2024-04-15T16:49:59Z","title":"LLM Evaluators Recognize and Favor Their Own Generations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13076","snapshot_observed_at":"2026-08-07T05:34:26.937721Z","title":"Bowman, and Shi Feng","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.937721Z"},"links":{"cited_paper":"/paper/2404.13076","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:0ce694fc20b2c2e3e69483a9802d167c18c1286f6d0682c74fb9a3cb9aa1a1c3","observation_id":"30ef8940-83f3-4a49-8f21-17c2e82ebbe2","resolution":{"observed_at":"2026-08-07T05:34:26.937721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-07T05:34:26.940907Z","title":"tinybenchmarks: evaluating llms with fewer examples","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.940907Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:4def5ef88f5ab9c5c5dca5278e39e21b041019a4f4acd483d0eea835029b40a2","observation_id":"7d31cfa5-4838-48d6-b3ba-5e84408948d3","resolution":{"observed_at":"2026-08-07T05:34:26.940907Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17202","last_updated":"2024-10-31T03:26:21Z","snapshot_observed_at":"2026-08-07T08:42:03.046490Z","submitted_at":"2024-05-27T14:24:47Z","title":"Efficient multi-prompt evaluation of LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17202","snapshot_observed_at":"2026-08-07T05:34:26.944282Z","title":"Efficient multi-prompt evaluation of llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.944282Z"},"links":{"cited_paper":"/paper/2405.17202","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:ebe803d564a2eaaafd4118a6b965ac9c1c88500236ab52596f2d69fbe83489c9","observation_id":"98b3f156-3235-4eb2-82b4-51d6389c886e","resolution":{"observed_at":"2026-08-07T05:34:26.944282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.668422Z","title":"SQuAD: 100,000+ questions for machine comprehension of text","venue":null,"work_id":"a44aa09c-376d-45f4-9873-5124cb3e091c","year":2016},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.947725Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:3875644acd9a8a14cc087f61bfb4120cbb40d3b3fbd2fee067a0357db4d929c3","observation_id":"c2570af8-3b21-4c0a-af04-5be2ba7ee595","resolution":{"observed_at":"2026-08-07T05:34:27.671738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-04T22:55:15.345443Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-07T05:34:26.950951Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.950951Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:c049ff6ff3238f2b2a94233c6d6b333609d15f9fe7c7e17a86f6ec3ca1d7c335","observation_id":"ed27291c-1724-485f-8517-1abc5e842edb","resolution":{"observed_at":"2026-08-07T05:34:26.950951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:26.954437Z","title":"Semiparametric efficiency in multivariate regression models with missing data","venue":null,"work_id":null,"year":1995},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.954437Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:fde70623e79de6ec2aaac986a10805c24993daa47298014385643d3a7d32708c","observation_id":"2f7c9bc9-ee91-4177-a4b8-dd0edfd42832","resolution":{"observed_at":"2026-08-07T05:34:26.954437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.652845Z","title":"Lalor, Robin Jia, and Jordan L","venue":null,"work_id":"5a563983-040f-43c1-84da-a3ddc4c8eaa4","year":2021},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.957403Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:31bcfc19d5759b1e72a9d4658bcbd22c2ac38a0009c5d42491cc1c1cca027653","observation_id":"2738fefe-3532-4e29-96a8-666d66c99118","resolution":{"observed_at":"2026-08-07T05:34:27.655964Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.10938","last_updated":"2024-10-01T23:38:10Z","snapshot_observed_at":"2026-07-06T18:15:54.402568Z","submitted_at":"2024-05-17T17:49:44Z","title":"Observational Scaling Laws and the Predictability of Language Model Performance","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.10938","snapshot_observed_at":"2026-08-07T05:34:26.960631Z","title":"Maddison, and Tatsunori B","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.960631Z"},"links":{"cited_paper":"/paper/2405.10938","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:78416ed843ddc0f7d1c1c59bcde87c09afbe2dde561971a039df5f76fafd3f5d","observation_id":"b515bf99-dcf7-4650-b11c-3290558b5a87","resolution":{"observed_at":"2026-08-07T05:34:26.960631Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.643143Z","title":"Bernstein, Alexander C","venue":null,"work_id":"c5484ec4-6844-425f-a38a-68af2a92b569","year":2014},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.963910Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:16995834f6b215ed49d72cf1b87bbf1e9f885ee54e441d5c11d48015e9ee965f","observation_id":"b60afbc5-1bd7-4340-ad53-90e5ba059210","resolution":{"observed_at":"2026-08-07T05:34:27.646469Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.633940Z","title":"Data distillation: A survey","venue":null,"work_id":"6a78a0de-4d2f-4d5a-a661-d7c14a8b24eb","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.966967Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:90f56e3862704f88fade67a68dec30338bd2a66ff5bbdbaedd24d1b688875881","observation_id":"6da416b0-5c7d-4d36-ae7e-bb0990390be9","resolution":{"observed_at":"2026-08-07T05:34:27.636832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-07T05:34:26.970380Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters.arXiv preprint arXiv:2408.03314, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.970380Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:f03eb1ae8327d625cc03e3fa894abef8ba8b615b04a5e2de4d145cb2f8203ddb","observation_id":"1e72658b-55a9-4ae9-a9f7-4698ca13d93c","resolution":{"observed_at":"2026-08-07T05:34:26.970380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.623777Z","title":"Recursive deep models for semantic compositionality over a sentiment treebank","venue":null,"work_id":"ff90db61-5ece-4e33-b099-77eaae674d4f","year":2013},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.973877Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:109961a33c51ec317224e8b2a941a450a73af04ef2799ecc4f26a624877aa8a2","observation_id":"97e17da9-1936-4277-9f66-41108ef856ea","resolution":{"observed_at":"2026-08-07T05:34:27.627625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16049","last_updated":"2024-03-23T21:21:44Z","snapshot_observed_at":"2026-08-04T21:08:49.454029Z","submitted_at":"2023-10-24T17:59:20Z","title":"MuSR: Testing the Limits of Chain-of-thought with Multistep Soft Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16049","snapshot_observed_at":"2026-08-07T05:34:26.976895Z","title":"Musr: Testing the limits of chain-of-thought with multistep soft reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.976895Z"},"links":{"cited_paper":"/paper/2310.16049","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:af5d9efbaa6f403192b1384f4ee1996297490ae4d7c8fb87bb72cb1ca50c23a5","observation_id":"0ffbcf20-b91c-480a-9863-65da0b63a71c","resolution":{"observed_at":"2026-08-07T05:34:26.976895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:26.980129Z","title":"Test-time training with self-supervision for generalization under distribution shifts","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.980129Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:fcecd0d06b1b36c550fc055b005782c8684344d42f790f89bc7a4b1b2d4f6b19","observation_id":"fad17933-bd87-4d4a-910f-86803c968dd7","resolution":{"observed_at":"2026-08-07T05:34:26.980129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.607575Z","title":"Le, Ed H","venue":null,"work_id":"a16e7e64-aab6-4b22-bc7b-68e50dbf1538","year":2022},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.983164Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:b457dccacecd0faa6c4b96633f7e87d803829c19efab12e140a3d01fddc5a426","observation_id":"a1446e7b-b1a7-4633-a1a4-c2523c4af7a0","resolution":{"observed_at":"2026-08-07T05:34:27.610789Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1612.06661","last_updated":"2017-11-04T21:37:30Z","snapshot_observed_at":"2026-07-06T05:23:29.184805Z","submitted_at":"2016-12-20T13:44:34Z","title":"Four lectures on probabilistic methods for data science","version":2},"cited_work":{"arxiv_id":"1612.06661","doi":null,"metadata_source":"pith","pith_arxiv_id":"1612.06661","snapshot_observed_at":"2026-08-07T05:34:27.138069Z","title":"Four lectures on probabilistic methods for data science","venue":"math.PR","work_id":"67daf342-27a3-4d8b-b7f0-eaaee84a964f","year":2016},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.986668Z"},"links":{"cited_paper":"/paper/1612.06661","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:ac8d0654134804b3c3bf7eecc0a0389844375d789c0067387da771ea3e29f269","observation_id":"6a3b525b-d889-45a6-8b7c-05356aa14d5e","resolution":{"observed_at":"2026-08-07T05:34:27.141439Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08638","last_updated":"2024-02-18T21:37:47Z","snapshot_observed_at":"2026-07-06T16:19:10.103540Z","submitted_at":"2023-09-14T17:45:51Z","title":"Anchor Points: Benchmarking Models with Much Fewer Examples","version":2},"cited_work":{"arxiv_id":"2309.08638","doi":null,"metadata_source":"pith","pith_arxiv_id":"2309.08638","snapshot_observed_at":"2026-08-07T05:34:27.123996Z","title":"Anchor Points: Benchmarking Models with Much Fewer Examples","venue":"cs.CL","work_id":"860145e2-904f-4ff9-b1a9-57ef64591049","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.989961Z"},"links":{"cited_paper":"/paper/2309.08638","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:43df36c0421fd1055f4af0bcd6860247208f3063a1630db0f907873481b3d1d6","observation_id":"5c68714b-37a1-4384-b026-c3e71eb8d677","resolution":{"observed_at":"2026-08-07T05:34:27.127869Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.598311Z","title":null,"venue":null,"work_id":"81846540-e2f6-4310-8fbe-0e2c227cb37f","year":2018},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.993144Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:03233f7c1caddcaa4da1e6a8075d9a8c1975269a0273bd7082e95b059ec65d63","observation_id":"ead71a23-d406-4f36-b537-835d2941fa44","resolution":{"observed_at":"2026-08-07T05:34:27.601129Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-07T05:34:26.996332Z","title":"Ku, Kai Wang, Alex Zhuang, Rongqi \"Richard\" Fan, Xiang Yue, and Wenhu Chen","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.996332Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:4bc767d9d7f78381874cfb3b9ef4ad14c3e2f65963ef8a4af10a0d2081b30384","observation_id":"aa39132f-b54c-4a1f-9498-b0cc5d5562c2","resolution":{"observed_at":"2026-08-07T05:34:26.996332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21819","last_updated":"2025-06-21T08:39:06Z","snapshot_observed_at":"2026-08-04T18:30:37.623897Z","submitted_at":"2024-10-29T07:42:18Z","title":"Self-Preference Bias in LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21819","snapshot_observed_at":"2026-08-07T05:34:26.999567Z","title":"Self-preference bias in llm-as-a-judge","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:26.999567Z"},"links":{"cited_paper":"/paper/2410.21819","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:b57130e7c14b7d365a10aac18876c6f46ac417ec0abd287c7e8d704018502c41","observation_id":"1b64fdac-d565-4747-8e86-db0fdbf64704","resolution":{"observed_at":"2026-08-07T05:34:26.999567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.588518Z","title":null,"venue":null,"work_id":"d7793ef6-f644-4bdf-ad1d-fe8f8206515f","year":2018},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.002653Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:f404d46d941a401bacc59eeb985188f40748f200ce8706f77131e1fa8aad974e","observation_id":"a11929c3-9425-4a9c-bf3c-74488c36aad4","resolution":{"observed_at":"2026-08-07T05:34:27.591844Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.17567","last_updated":"2023-10-26T16:55:05Z","snapshot_observed_at":"2026-08-06T20:11:27.128480Z","submitted_at":"2023-10-26T16:55:05Z","title":"Skill-Mix: a Flexible and Expandable Family of Evaluations for AI models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.17567","snapshot_observed_at":"2026-08-07T05:34:27.005749Z","title":"Skill-mix: a flexible and expandable family of evaluations for ai models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.005749Z"},"links":{"cited_paper":"/paper/2310.17567","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:c70ecc8e1e5691350099b78512ab9d0ae543acf8245f8410c5c32f1f82583f66","observation_id":"18b48d55-7d95-4d69-8dad-1b3795ef9cab","resolution":{"observed_at":"2026-08-07T05:34:27.005749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.578601Z","title":"Automatic evaluation of attribution by large language models","venue":null,"work_id":"fc007b91-fb64-40a8-97b4-976d1d2826eb","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.008828Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:954e8e3b1183d1ab3dd307b997ed93983702e76eefd777904ae56789c3efdc5d","observation_id":"dff6138b-4a97-4362-b2f6-3655a56c0bba","resolution":{"observed_at":"2026-08-07T05:34:27.582423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-08-07T05:34:27.011892Z","title":"Instruction-following evaluation for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.011892Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:d1b63c38178d077383aac9efa5e351a36681bda395db003bfb2adf1576b7ef13","observation_id":"9cbd5450-d806-4bfd-8b28-5ab478823d84","resolution":{"observed_at":"2026-08-07T05:34:27.011892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06172","last_updated":"2025-02-26T21:53:59Z","snapshot_observed_at":"2026-08-01T22:09:47.848578Z","submitted_at":"2024-07-08T17:48:42Z","title":"On Speeding Up Language Model Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06172","snapshot_observed_at":"2026-08-07T05:34:27.015178Z","title":"Belardi, Ruihan Wu, Travis Zhang, Carla P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.015178Z"},"links":{"cited_paper":"/paper/2407.06172","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:4bc3980b2681947a9158f697a92a3661e4b05220eedd1cc0a252d6479584c778","observation_id":"e8a454e4-155c-44a6-93ef-63ffc29a4e1c","resolution":{"observed_at":"2026-08-07T05:34:27.015178Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.09880","last_updated":"2023-01-24T09:37:00Z","snapshot_observed_at":"2026-08-06T00:08:18.875420Z","submitted_at":"2023-01-24T09:37:00Z","title":"Probabilistic Bilevel Coreset Selection","version":1},"cited_work":{"arxiv_id":"2301.09880","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.09880","snapshot_observed_at":"2026-08-07T05:34:27.053613Z","title":"Probabilistic Bilevel Coreset Selection","venue":"cs.LG","work_id":"911529e1-a091-429d-a81d-7969de4182d7","year":2023},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.018282Z"},"links":{"cited_paper":"/paper/2301.09880","citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:e73a13001dd852f40ba2b0ad73f26c1bba4adab470b73bb5d1b67f5c10681fba","observation_id":"8e104514-0e04-4d17-a20c-fc8609e6259e","resolution":{"observed_at":"2026-08-07T05:34:27.059462Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T05:34:27.569221Z","title":"How to select datapoints for efficient human evaluation of nlg models?, 2025","venue":null,"work_id":"63deaa5c-959a-4c40-8dbf-3e0b72dc5b3b","year":2025},"citing_paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark","version":2},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T05:34:27.021795Z"},"links":{"citing_paper":"/paper/2506.07673"},"observation_digest":"sha256:ac0a66555425d7ade81c4e4d8fbf105946336bf0416ebe980f5c43e450f5145b","observation_id":"1e3e299b-9295-4d96-8c3f-399e60260ee5","resolution":{"observed_at":"2026-08-07T05:34:27.572476Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.07673","last_updated":"2026-07-21T11:15:43Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T05:26:00.127524Z","submitted_at":"2025-06-09T11:50:41Z","title":"How Benchmark Prediction from Fewer Data Misses the Mark"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":6,"verified_fuzzy":21},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 6 inbound Pith citation observations for arXiv:2506.07673."}