{"as_of":"2026-08-08T20:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cabd11a5a1a07aff43a686efb8009238d8efc6514c8bb51ff504363a6db59981","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:50:35.589753Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.09813/citation-record","integrity":"/paper/2506.09813/integrity","json":"/paper/2506.09813/citation-record.json","paper":"/paper/2506.09813"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:37.713801Z","title":"Suppose we have a setKsuch that|K| ≤(α−1)gandKsatisfies positional representation for group sizeg","venue":null,"work_id":"d84cbc48-a6f3-4997-9018-8f47289778c9","year":2015},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.100847Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:8a05fbb2d595274dd8f4504d762a671059f9d64acf0a7abc310853c149bac198","observation_id":"c57aac52-22a9-4036-9b65-d8c1d6e03992","resolution":{"observed_at":"2026-08-07T04:50:37.858463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:38.026971Z","title":"coalition","venue":null,"work_id":"f0d17530-4454-4f9a-b7d6-813af40d86ed","year":2017},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.987732Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:bd3ea0336a221c75990c4b28cb8a77bce95c82b0f26cd7911b5b12f20db1b82e","observation_id":"fe0f85e5-9004-42af-ac06-bec14899dcee","resolution":{"observed_at":"2026-08-07T04:50:38.166042Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.269088Z","title":"ex- isting subset","venue":null,"work_id":"2cff7daa-b14c-48c0-887c-f36cb64b329b","year":2025},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.589753Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:15d164356a74a32e8f5c1615276e7b63fae337d5d1a3f63818074ae89a0cfc33","observation_id":"179b307d-bd0b-44c6-ab8b-2a97d59233f7","resolution":{"observed_at":"2026-08-07T04:50:36.338871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:37.158243Z","title":"exact cover by 3 sets","venue":null,"work_id":"ac00432d-fc96-4c19-a66d-9647c45abb63","year":1995},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.294832Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:2fa83b1e81220118c365ffd04a7ab91c01cae0eb597aff58e115ba2d43fa1431","observation_id":"4b06efb5-dea8-41df-aa47-6503a6e0b70a","resolution":{"observed_at":"2026-08-07T04:50:37.345915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.908084Z","title":"Big-G”, “Big-G sparse","venue":null,"work_id":"5827c60f-290f-4734-bcbf-8d364f890a99","year":2022},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.364011Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:5a140da8820e145e1944fa238b85c1010d0b38699e668a28298cf71096864b5f","observation_id":"f49c6c56-08db-4802-bab1-524e6683d252","resolution":{"observed_at":"2026-08-07T04:50:37.045949Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.05769","last_updated":"2023-02-12T13:02:27Z","snapshot_observed_at":"2026-08-03T00:51:49.109121Z","submitted_at":"2022-10-11T20:19:11Z","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","version":3},"cited_work":{"arxiv_id":"2210.05769","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.05769","snapshot_observed_at":"2026-08-07T04:50:35.794918Z","title":"Vote'n'Rank: Revision of Benchmarking with Social Choice Theory","venue":"cs.LG","work_id":"08d9f016-7a06-4b02-b1c5-e67bd02fb391","year":2022},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.653219Z"},"links":{"cited_paper":"/paper/2210.05769","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:86d099ff3dee0dc86ec6303044bd91969568f080dd38a7b9eb1491a51aa2a77e","observation_id":"ae2b18e2-2a99-4110-a9ab-944c2924bd15","resolution":{"observed_at":"2026-08-07T04:50:35.843606Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:37.460322Z","title":"Finally, we can conclude that there mustexist some rankingσ N such that noKwith size|K| ≤ 1 288ϵ2 log(m) satisfiesϵ-positional proportionality","venue":null,"work_id":"350b4b67-0ef4-4c38-aab5-5719871e5a49","year":2001},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.229886Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:2372f678bf713b145db16767a6e2a876c5c07f3492b974f5605ca7119d19a600","observation_id":"e7dcb8e8-276b-49b0-97d9-265862461d9f","resolution":{"observed_at":"2026-08-07T04:50:37.595715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.719897Z","title":"representative","venue":null,"work_id":"15e8f606-c9ea-4333-9302-8e88305d563b","year":2022},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.456242Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:49bdcc5ae4218dbf903011d37b422e72b9680c30c3dca79162e8bb0ff925a6c5","observation_id":"2e825802-02be-4ba3-9f29-68bb5d52dea7","resolution":{"observed_at":"2026-08-07T04:50:36.809215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:36.419477Z","title":"Core scenarios","venue":null,"work_id":"f870c230-abb7-4c37-aa3e-a56b65b5dc4c","year":2025},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:35.520700Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:b9ef698262b2c261842af297fedef6607dc6fd2ef45a15983a8f8bb0f0b75717","observation_id":"f0529063-da86-4bf5-8efc-b79521ae2bcb","resolution":{"observed_at":"2026-08-07T04:50:36.558221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12015","last_updated":"2025-01-21T10:13:28Z","snapshot_observed_at":"2026-07-06T20:23:45.022640Z","submitted_at":"2025-01-21T10:13:28Z","title":"Full Proportional Justified Representation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12015","snapshot_observed_at":"2026-08-07T04:50:33.898778Z","title":"Full proportional justified representa- tion.arXiv preprint arXiv:2501.12015,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1973,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.898778Z"},"links":{"cited_paper":"/paper/2501.12015","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:73975ffae416269d563dd7c00f56c9adba8a1f32795244df0f1874031cb40227","observation_id":"e038a36f-f85a-4fa4-8f33-cd1fb85fdadf","resolution":{"observed_at":"2026-08-07T04:50:33.898778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01719","last_updated":"2024-05-06T15:09:50Z","snapshot_observed_at":"2026-07-06T18:09:04.820856Z","submitted_at":"2024-05-02T20:28:54Z","title":"Inherent Trade-Offs between Diversity and Stability in Multi-Task Benchmarks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01719","snapshot_observed_at":"2026-08-07T04:50:34.884507Z","title":"Inherent trade-offs between diversity and stability in multi-task benchmark.arXiv preprint arXiv:2405.01719,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1975,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.884507Z"},"links":{"cited_paper":"/paper/2405.01719","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:dfb6efea9f881d75cad5a1f531fb5ad4f349080e570e1355812c474002e6ce4a","observation_id":"dc3a456d-22d5-4e9f-8c31-d09245a1dd4d","resolution":{"observed_at":"2026-08-07T04:50:34.884507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.10284","last_updated":"2023-05-17T15:20:31Z","snapshot_observed_at":"2026-07-06T15:28:44.711559Z","submitted_at":"2023-05-17T15:20:31Z","title":"Towards More Robust NLP System Evaluation: Handling Missing Scores in Benchmarks","version":1},"cited_work":{"arxiv_id":"2305.10284","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.10284","snapshot_observed_at":"2026-08-07T04:50:36.132905Z","title":"Towards More Robust NLP System Evaluation: Handling Missing Scores in Benchmarks","venue":"cs.CL","work_id":"4fb01280-ae84-4c8a-ad11-c0f056bb1c60","year":2023},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1979,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.571425Z"},"links":{"cited_paper":"/paper/2305.10284","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:98e216fdd13513ee52d867483cd758219aad0b1b3c8f9a261889a2b636b08a3d","observation_id":"c3b10293-3d5f-4ccf-ba46-169950cbb4e1","resolution":{"observed_at":"2026-08-07T04:50:36.186825Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.11696","last_updated":"2024-04-01T17:34:34Z","snapshot_observed_at":"2026-07-06T16:09:12.888489Z","submitted_at":"2023-08-22T17:59:30Z","title":"Efficient Benchmarking of Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.11696","snapshot_observed_at":"2026-08-07T04:50:34.399951Z","title":"Efficient benchmarking of lan- guage models.arXiv preprint arXiv:2308.11696,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":1995,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.399951Z"},"links":{"cited_paper":"/paper/2308.11696","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:1222700c412a3d298ec54dafa842afd54c13425126d5cdd019c8c4364215139d","observation_id":"c8e958b9-687b-4b4d-9dad-93e531d2c948","resolution":{"observed_at":"2026-08-07T04:50:34.399951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:38.330758Z","title":"Proportionally fair clustering revisited","venue":null,"work_id":"f88e80d7-6d35-4e58-93b5-7527d23a028a","year":2020},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2001,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.258254Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:1f39e0f8bc0667ea9c99563d2c4d52594c7bbd8e153c9b4145ab44afddc6f5c6","observation_id":"8baec2d6-d329-49a3-943c-d75d13ba6cab","resolution":{"observed_at":"2026-08-07T04:50:38.470912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-07T04:50:33.759739Z","title":"Swe-bench: Can language models resolve real-world github issues? arXiv preprint arXiv:2310.06770,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2009,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.759739Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:2c596b66bd409e1305550c45527f752a8c93ea66183e38b4b279e9ed64baa68b","observation_id":"6925d52e-4988-4184-b348-c0273ccba638","resolution":{"observed_at":"2026-08-07T04:50:33.759739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-07T04:50:34.762614Z","title":"Beyond the imitation game: Quantifying and extrapolating the capabilities of language models.arXiv preprint arXiv:2206.04615,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.762614Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:7029f4762048dc367b6bc49b2d0da11f45a7d1638ee8d16da340d789f0ee8108","observation_id":"6b5578c3-4049-4f60-bfec-3e11530af946","resolution":{"observed_at":"2026-08-07T04:50:34.762614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-05T12:01:06.707349Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-07T04:50:34.511713Z","title":"tinybenchmarks: evaluating llms with fewer examples.arXiv preprint arXiv:2402.14992,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.511713Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:47ecd849bb6d134592ff4caee28951918d4be833c46cad74751842d919834fe0","observation_id":"a856b101-903c-4d8b-b96d-34f9f7d54854","resolution":{"observed_at":"2026-08-07T04:50:34.511713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T04:50:38.592258Z","title":"Exact algorithms for set multicover and multiset multicover problems","venue":null,"work_id":"d8b7bd71-7fc7-42b9-9013-1e7cd6c04f18","year":2009},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:33.655898Z"},"links":{"citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:04b388e858ca9a96b5947bcd38568ab7144d4cf395fd79e7e9e654f37877a121","observation_id":"1e67bd58-e739-47c0-9caa-5c6f1e8dd4bd","resolution":{"observed_at":"2026-08-07T04:50:38.685753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-07T04:50:34.163425Z","title":"Holistic evaluation of language models.arXiv preprint arXiv:2211.09110,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.163425Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:4d9b72e1da3b19d43ca0fbf9c1f0f0fbff62bd643f4ed402381c931d8fc12b39","observation_id":"45d531a0-0a4f-4002-ba70-99ca07e15346","resolution":{"observed_at":"2026-08-07T04:50:34.163425Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05952","last_updated":"2024-10-08T12:08:46Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:08:46Z","title":"Active Evaluation Acquisition for Efficient LLM Benchmarking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05952","snapshot_observed_at":"2026-08-07T04:50:34.063035Z","title":"Active evaluation acquisition for efficient llm benchmarking.arXiv preprint arXiv:2410.05952,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-07T04:50:34.063035Z"},"links":{"cited_paper":"/paper/2410.05952","citing_paper":"/paper/2506.09813"},"observation_digest":"sha256:43789556d04f06188732f4617d2c05ded2aad226deb23a2ceeffdf989d77a49b","observation_id":"5146f728-8393-40fd-9c30-057814cb016a","resolution":{"observed_at":"2026-08-07T04:50:34.063035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2506.09813","last_updated":"2025-06-16T12:43:27Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T04:37:22.028673Z","submitted_at":"2025-06-11T14:53:47Z","title":"Metritocracy: Representative Metrics for Lite Benchmarks"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 20 of 20 outbound references and 0 inbound Pith citation observations for arXiv:2506.09813."}