{"as_of":"2026-08-10T22:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c0280a3da4db6f8fea5a16a32e9c65065aa2f2542f24767096a46141f3c326fb","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T21:43:06.616805Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T23:34:30.024509Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T23:47:27.677181Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04234","snapshot_observed_at":"2026-08-04T23:34:30.024509Z","title":"Statistical uncertainty quan- tification for aggregate performance metrics in machine learning benchmarks","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.06472","last_updated":"2025-09-09T08:54:11Z","snapshot_observed_at":"2026-08-09T19:12:00.821489Z","submitted_at":"2025-09-08T09:37:20Z","title":"Rethinking LLM Parametric Knowledge as Post-retrieval Confidence for Dynamic Retrieval and Reranking","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T23:34:30.024509Z"},"links":{"cited_paper":"/paper/2501.04234","citing_paper":"/paper/2509.06472"},"observation_digest":"sha256:b80d2496f0d2d213c6b082559939e94eb54e5bdbc0a96203f09a91d3457fb5f1","observation_id":"665aba2b-438d-435f-a8da-e66539ee39eb","resolution":{"observed_at":"2026-08-04T23:34:30.024509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"cited_work":{"arxiv_id":"2501.04234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.04234","snapshot_observed_at":"2026-07-02T23:47:27.677181Z","title":"org/abs/2501.04234","venue":null,"work_id":"a04cc81f-77b0-4b9c-93e1-d384288984c1","year":2025},"citing_paper":{"arxiv_id":"2604.23102","last_updated":"2026-04-25T01:55:54Z","snapshot_observed_at":"2026-08-08T00:20:29.964910Z","submitted_at":"2026-04-25T01:55:54Z","title":"Unstable Rankings in Bayesian Deep Learning Evaluation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-08T08:34:44.254637Z"},"links":{"cited_paper":"/paper/2501.04234","citing_paper":"/paper/2604.23102"},"observation_digest":"sha256:8122e6c11697021c37b95e97dd481f87eb8864dbbd666d74593fd9455dad8191","observation_id":"d087617a-4524-4cd8-a690-bc1ca6097709","resolution":{"observed_at":"2026-05-11T20:31:15.186122Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"cited_work":{"arxiv_id":"2501.04234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.04234","snapshot_observed_at":"2026-07-02T23:47:27.677181Z","title":"org/abs/2501.04234","venue":null,"work_id":"a04cc81f-77b0-4b9c-93e1-d384288984c1","year":2025},"citing_paper":{"arxiv_id":"2604.23114","last_updated":"2026-04-25T02:52:33Z","snapshot_observed_at":"2026-07-06T23:09:23.591544Z","submitted_at":"2026-04-25T02:52:33Z","title":"A Tale of Two Variances: When Single-Seed Benchmarks Fail in Bayesian Deep Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-08T08:26:01.717280Z"},"links":{"cited_paper":"/paper/2501.04234","citing_paper":"/paper/2604.23114"},"observation_digest":"sha256:54ba4d1760f12b9fc7e4454f632b3c4dd8861f6624fa551ce5e022d06b8efa74","observation_id":"83b60adc-67a0-46a5-a3fc-5402f697f8ad","resolution":{"observed_at":"2026-05-11T20:36:11.788732Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"cited_work":{"arxiv_id":"2501.04234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.04234","snapshot_observed_at":"2026-07-02T23:47:27.677181Z","title":"org/abs/2501.04234","venue":null,"work_id":"a04cc81f-77b0-4b9c-93e1-d384288984c1","year":2025},"citing_paper":{"arxiv_id":"2606.08679","last_updated":"2026-06-07T15:31:29Z","snapshot_observed_at":"2026-08-04T23:05:45.103241Z","submitted_at":"2026-06-07T15:31:29Z","title":"Rank Intervals for Leaderboards: A Hierarchical Framework for Model Evaluation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T17:54:22.974336Z"},"links":{"cited_paper":"/paper/2501.04234","citing_paper":"/paper/2606.08679"},"observation_digest":"sha256:b8d82ec3fc1de03551359ee7df8932d504f4f3b41d1f3f19c0f29d6a951e275f","observation_id":"c3a5bc77-6ef5-4ef8-bb6c-936e0b756488","resolution":{"observed_at":"2026-07-02T23:47:27.679532Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.04234","snapshot_observed_at":"2026-08-02T09:45:26.760955Z","title":"Statistical uncertainty quantification for aggregate performance met- rics in machine learning benchmarks.arXiv preprint arXiv:2501.04234,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16259","last_updated":"2026-06-28T11:19:02Z","snapshot_observed_at":"2026-08-10T21:46:57.012531Z","submitted_at":"2026-06-28T11:19:02Z","title":"Quantifying Ranking Uncertainty in LLM Benchmarks","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-02T09:45:26.760955Z"},"links":{"cited_paper":"/paper/2501.04234","citing_paper":"/paper/2607.16259"},"observation_digest":"sha256:5e63c0d94cfdfc512c239031bf782ee0f2e1fe84ef7875acefbd2fb79d80a65a","observation_id":"3514908c-2504-42d2-9ea3-a515cef37245","resolution":{"observed_at":"2026-08-02T09:45:26.760955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.04234/citation-record","integrity":"/paper/2501.04234/integrity","json":"/paper/2501.04234/citation-record.json","paper":"/paper/2501.04234"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-10T21:43:06.499332Z","title":"GPT -4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.499332Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:9cf5534fd5183689821b4c04a94d08346635e7404a1d550273930e85bc382cae","observation_id":"3ee6cec1-6a14-4895-9c83-9fe2550131a8","resolution":{"observed_at":"2026-08-10T21:43:06.499332Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:07.002839Z","title":"Bayesian inferences on uncertain ranks and orderings: Application to ranking players and lineups","venue":null,"work_id":"d81b4970-d583-47e3-8fa7-34eba58350b5","year":2023},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.504544Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:39a57a381f99c4c2d7c725c43d096072c01c8f220ef425e2d89408f31de51aff","observation_id":"6ce775bc-e399-466f-87b3-61d0fddea82c","resolution":{"observed_at":"2026-08-10T21:43:07.007179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.990265Z","title":"Time for a change: A tutorial for comparing multiple classifiers through Bayesian analysis","venue":null,"work_id":"6aa85adb-c560-4dd9-aadc-87d41079ff45","year":2017},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.508818Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:346d0b82d78a39e28ae89b7e8440c68e65a64b9890b1da98ecf3e4f580a87739","observation_id":"4212ce72-189f-44e7-b5af-716038db8c01","resolution":{"observed_at":"2026-08-10T21:43:06.994702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.978881Z","title":"Hudson, Ehsan Adeli, Russ Altman, Simran Arora, Sydney von Arx, Michael S","venue":null,"work_id":"c7affa46-ccd8-4461-88fb-b02ded7219bd","year":2021},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.513317Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:4ae04c14de6930a53e8882c31635335f90e533721e86495d2065e0d69bb94cbb","observation_id":"c040c238-9c07-47ee-85a6-7f3826dee3b5","resolution":{"observed_at":"2026-08-10T21:43:06.982741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.965827Z","title":"Accounting for variance in machine learning benchmarks","venue":null,"work_id":"728505bd-6492-4572-8c20-0e6e76a58455","year":2021},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.517895Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:cfeb687372a867451dbb9121da63c08e9817e887886554649b13e20ff1e94e94","observation_id":"b24df635-f720-4d26-88fc-70487398d3e6","resolution":{"observed_at":"2026-08-10T21:43:06.970718Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.953458Z","title":"What are the best systems? N ew perspectives on NLP benchmarking","venue":null,"work_id":"6f602698-7ee1-4071-a795-b345f2277c32","year":2022},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.522408Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:ceebea53162a88c22f8737df7cc4a8d2e3eb54008303c2845d5719001c8ad2bd","observation_id":"5bce9be4-2f6d-4392-92b4-75330a7ec61e","resolution":{"observed_at":"2026-08-10T21:43:06.957749Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.07002","last_updated":"2021-07-14T21:08:30Z","snapshot_observed_at":"2026-07-06T11:29:14.236629Z","submitted_at":"2021-07-14T21:08:30Z","title":"The Benchmark Lottery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.07002","snapshot_observed_at":"2026-08-10T21:43:06.526759Z","title":"The benchmark lottery","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.526759Z"},"links":{"cited_paper":"/paper/2107.07002","citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:8fd83ed490d291221fb73be32a1307c2d8e343ee6015932e5c1dfab2de00d3e4","observation_id":"3c0f5b98-4385-4e21-9fbf-bbdac12fa03b","resolution":{"observed_at":"2026-08-10T21:43:06.526759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.942016Z","title":"Statistical comparisons of classifiers over multiple data sets","venue":null,"work_id":"dcffc2c0-4546-4b3f-abfc-308a8d78cd8a","year":2006},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.530628Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:fe89d688fabe14578a65a8a66f01662700905c7f85657414957cd2847a0543b3","observation_id":"86d6398b-416a-41b6-96a4-69e0c5ef7c5f","resolution":{"observed_at":"2026-08-10T21:43:06.945859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.929577Z","title":"Bayesian aggregation of order-based rank data","venue":null,"work_id":"e8e7f8e3-42c7-4c80-a1f1-44f21664e61d","year":2014},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.534246Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:2d846514f466ff19b3f657f8f94f9bcbf3edab2279cddfb4fda402acde2e374f","observation_id":"2ae27b53-2cea-4c6a-a380-6436a2d15947","resolution":{"observed_at":"2026-08-10T21:43:06.933812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.916567Z","title":"Approximate statistical tests for comparing supervised classification learning algorithms","venue":null,"work_id":"2e0f887a-4d4a-46d8-b905-2d5040df2a4b","year":1923},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.538052Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:1157a6e67bfc671ef945cdc101cd323f548952f122c04d46d450d9ab4bbee435","observation_id":"afae105b-2648-43b3-b27e-d32223867a54","resolution":{"observed_at":"2026-08-10T21:43:06.920637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.904409Z","title":"Statistical significance testing for natural language processing","venue":null,"work_id":"927ed57c-c7a2-4af7-bd89-8263ec60348a","year":2020},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.541830Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:16502ab70c7f927fb4be6e1ae45f5900926df9cc7e15915a81c99c41952613be","observation_id":"fa0b86bb-d71f-46a2-9cf1-52fcf4c1d1a3","resolution":{"observed_at":"2026-08-10T21:43:06.908376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.545426Z","title":"An Introduction to the Bootstrap","venue":null,"work_id":null,"year":1994},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.545426Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:5ca1ec743c8dcd9aeab08a7bd18fba4e3de116ebf4c1af64803868bee6e10127","observation_id":"a03c59fb-ff64-451b-b43c-6be47a494543","resolution":{"observed_at":"2026-08-10T21:43:06.545426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.885554Z","title":"The graphical presentation of a collection of means","venue":null,"work_id":"f50dec98-a02a-4c70-b578-409dca94327c","year":1995},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.549358Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:ab5bb1f7acedc505d3c323c83adff98fc46f25aeca6cdf1d156bbe9151ce6584","observation_id":"6cf60cdd-e389-4cdb-9b7b-6d4f3e37775b","resolution":{"observed_at":"2026-08-10T21:43:06.889354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.874281Z","title":"League tables and their limitations: S tatistical issues in comparisons of institutional performance","venue":null,"work_id":"7f42efeb-eed2-4dc7-ac1f-557855b5433d","year":1996},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.552950Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:802bf54e08a2db60dce1a28217ab0ed6505851fcf342b8e47999bdee14aa4dc4","observation_id":"eca38a4f-76de-4952-b108-f88de834458d","resolution":{"observed_at":"2026-08-10T21:43:06.878323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.861636Z","title":"Randomized significance tests in machine translation","venue":null,"work_id":"c6907a6f-51f5-423b-b462-a19dd23d797b","year":2014},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.556318Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:64a7c09c201367378bf062a1211fbc0dd88bc926b5be926b00d2415bcd0030b9","observation_id":"d7da6371-c1dc-4d90-9473-3a247f0731aa","resolution":{"observed_at":"2026-08-10T21:43:06.866528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.849269Z","title":"Modeling the variability of rankings","venue":null,"work_id":"b64be472-a781-4ec5-bfe9-781513df61e7","year":2010},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.559927Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:c4dbf66406d63ebb34f5b084a4508f5454d54bc3ae5f2ffa93b1d4885b3c61ba","observation_id":"651d2b0e-5c5b-4fd7-8a29-f72b5187211c","resolution":{"observed_at":"2026-08-10T21:43:06.853267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.836309Z","title":"Statistical comparisons of classifiers by generalized stochastic dominance","venue":null,"work_id":"b97c6b82-2653-49df-9c7b-eb697630909a","year":2023},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.563813Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:0936da3578d0842337f92800a9bde34a379f582b983cdd65534eb19ca0248174","observation_id":"2e5a9dd2-22fb-4e09-b1a6-6779713d3a43","resolution":{"observed_at":"2026-08-10T21:43:06.841019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.823055Z","title":"Active Bayesian assessment of black-box classifiers","venue":null,"work_id":"6fb9b14a-84b7-4049-ad71-bf20663ea2b4","year":2021},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.567510Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:905b98678e4a78885a86bf136e22049f9a62914d300ef8975138cb59b5fa53ac","observation_id":"123283bf-cb14-447c-a32c-e9a9b52625be","resolution":{"observed_at":"2026-08-10T21:43:06.827661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.810570Z","title":"Theory of Point Estimation","venue":null,"work_id":"da072504-d28e-44df-af87-3d03322e9787","year":1998},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.571098Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:ebeb7371bc67b527af16cf430a128844d8353f4bbae6cdef3dc36d08857161a8","observation_id":"0c55c288-6523-4424-abc1-446d5bc5ca62","resolution":{"observed_at":"2026-08-10T21:43:06.814772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.798038Z","title":"Bayesian analysis of rank data with covariates and heterogeneous rankers","venue":null,"work_id":"9234ba01-39e5-4585-8ca7-ab6d45e10fef","year":2022},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.574833Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:d9938404443f5cdbc7a060e10b3c9f67707a42df0f5c1f19b8acdbce7427c1a5","observation_id":"acdeee00-44d8-4ef0-b3d4-c74dbc1a7e7b","resolution":{"observed_at":"2026-08-10T21:43:06.802155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.579923Z","title":"Slice sampling","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.579923Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:3c2365fb558d2b9485f2faa15fddaf023978704a5b3592dac1ce50bcc583a5f3","observation_id":"000e0b1b-144c-424f-8db9-63accd29a40c","resolution":{"observed_at":"2026-08-10T21:43:06.579923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03459","last_updated":"2023-06-21T03:03:15Z","snapshot_observed_at":"2026-08-10T20:34:08.641596Z","submitted_at":"2021-07-07T19:41:11Z","title":"Uncertainty in Ranking","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03459","snapshot_observed_at":"2026-08-10T21:43:06.583624Z","title":"Uncertainty in ranking","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.583624Z"},"links":{"cited_paper":"/paper/2107.03459","citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:6bf2ce205b98edbe890d2a55e9354a5bae760eaea365355a3d09939675abc45d","observation_id":"8dca4664-9298-47f1-83da-5451cd1b1a9e","resolution":{"observed_at":"2026-08-10T21:43:06.583624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.778565Z","title":"On comparing classifiers: Pitfalls to avoid and a recommended approach","venue":null,"work_id":"c4360386-dd0b-4a27-a30e-48c51f6fc806","year":1997},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.587733Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:0348c7b2e9508688f548640b0315158ae68dc3944536da6a679307a712a42379","observation_id":"730be737-3b13-4fff-97fa-a7ed45edcdc4","resolution":{"observed_at":"2026-08-10T21:43:06.782703Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.5281/zenodo.1068996","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.646003Z","title":null,"venue":null,"work_id":"7148817e-a0a7-4f07-b1d5-f02cc5250f97","year":2017},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.591539Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:b5e8436819af8679cbf52a3c71f04cfd6822ff57f5bfbe0c2f840fb5cb833cfa","observation_id":"2790ecde-d15d-4e1c-b6a3-1a9f61e32af1","resolution":{"observed_at":"2026-08-10T21:43:06.651434Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-10T21:43:06.595642Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.595642Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:3d92358519ddaeec889cb857ce9bcbd6d9bda6b478ef09f75ce882b0eb1322c2","observation_id":"a8292bee-30c6-4278-a8ea-7e7cb8bbfc9e","resolution":{"observed_at":"2026-08-10T21:43:06.595642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.765920Z","title":"Classifier uncertainty: E vidence, potential impact, and probabilistic treatment","venue":null,"work_id":"ab0bcd50-fa62-441b-b4fd-d247e00a64d9","year":2021},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.600120Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:6737859307a9fc18f4cae239cb55838571d7d4d794840d89f032fb77d25851c8","observation_id":"aafa2a5d-bd87-4698-895c-8ba22576d0b3","resolution":{"observed_at":"2026-08-10T21:43:06.770303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-10T21:43:06.604550Z","title":"LLaMA : Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.604550Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:99d10fc1d4c58eb83dd2ef3e79af05e89335f8f7b8f2b4ed835ab3c9df8d64bd","observation_id":"56df0d95-102f-4570-a73d-4924f262a84b","resolution":{"observed_at":"2026-08-10T21:43:06.604550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.753284Z","title":"Follow the leader (board) with confidence: E stimating p-values from a single test set with item and response variance","venue":null,"work_id":"a9ce7820-fdb4-448c-8114-3c3f3f8c8819","year":2023},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.608568Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:1c6434343e1454960a7c86012c72ad74a3ed3a76f6b2a4fb2f2d41d789595e48","observation_id":"6edff260-ef13-4ad0-9033-af101c7eba67","resolution":{"observed_at":"2026-08-10T21:43:06.757704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T21:43:06.736862Z","title":"Confidence intervals for population ranks in the presence of ties and near ties","venue":null,"work_id":"6973341e-b1a8-4f69-87f3-133c02b71bd9","year":2009},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.612790Z"},"links":{"citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:53376763d5a2d950f533da7196e8e167da5bbc06c47246ed13e33c7f6b889d8d","observation_id":"86801887-c27a-478a-b0c3-9e707c3d3173","resolution":{"observed_at":"2026-08-10T21:43:06.744142Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.04867","last_updated":"2020-02-21T13:36:15Z","snapshot_observed_at":"2026-08-09T06:48:42.729935Z","submitted_at":"2019-10-01T17:06:29Z","title":"A Large-scale Study of Representation Learning with the Visual Task Adaptation Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.04867","snapshot_observed_at":"2026-08-10T21:43:06.616805Z","title":"A large-scale study of representation learning with the visual task adaptation benchmark","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-10T21:43:06.616805Z"},"links":{"cited_paper":"/paper/1910.04867","citing_paper":"/paper/2501.04234"},"observation_digest":"sha256:a15ba3c656af98d293b636a7b999d7b7259ae292ee6987daaa1cdd77dd2520e5","observation_id":"de1b81ab-b83a-4acf-955c-cdf7dd101341","resolution":{"observed_at":"2026-08-10T21:43:06.616805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2501.04234","last_updated":"2025-01-08T02:17:34Z","latest_version":1,"primary_category":"stat.ML","snapshot_observed_at":"2026-08-10T21:44:43.433549Z","submitted_at":"2025-01-08T02:17:34Z","title":"Statistical Uncertainty Quantification for Aggregate Performance Metrics in Machine Learning Benchmarks"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":21},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 5 inbound Pith citation observations for arXiv:2501.04234."}