{"as_of":"2026-08-08T15:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:518a483ecad6283593ba946c48ac1dd8e25fe1a7693262afac2bd5a6f946f57e","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T11:47:48.971268Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T18:40:50.139345Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2501.14249"},"observation_digest":"sha256:df80518d2afbb7600a4d8e0e3efdaae63a8224ec6b42b310e2f8b8ef41149fa9","observation_id":"39a34301-2c01-4c36-8d7d-52060d5d47d8","resolution":{"observed_at":"2026-05-10T18:40:50.375124Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-08T11:47:48.971268Z","title":"Dynabench: Rethinking benchmarking in nlp","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.07732","last_updated":"2025-06-07T19:57:55Z","snapshot_observed_at":"2026-08-08T11:43:17.920153Z","submitted_at":"2025-02-11T17:51:52Z","title":"When Incentives Backfire, Data Stops Being Human","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-08T11:47:48.971268Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2502.07732"},"observation_digest":"sha256:9140638040f8f86236c8d54088923297d86c70fa0e2bd6ea8906caaadc1441b5","observation_id":"4b504f72-8c3c-41cc-9278-205d85bd7d02","resolution":{"observed_at":"2026-08-08T11:47:48.971268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-07T22:21:52.595165Z","title":"Dynabench: Rethinking benchmarking in nlp","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.09192","last_updated":"2025-05-27T17:24:38Z","snapshot_observed_at":"2026-08-07T22:17:35.075162Z","submitted_at":"2025-02-13T11:32:09Z","title":"Thinking beyond the anthropomorphic paradigm benefits LLM research","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T22:21:52.595165Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2502.09192"},"observation_digest":"sha256:7f1ad63508da0077f5898694495568aa64754036ef67ab0dd2ada18c1d4d9b68","observation_id":"3b929376-432b-4282-9211-7a2ad3be4b8f","resolution":{"observed_at":"2026-08-07T22:21:52.595165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-07T12:45:41.067226Z","title":"Dynabench: Rethinking benchmarking in nlp,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.23598","last_updated":"2025-05-29T16:11:18Z","snapshot_observed_at":"2026-08-07T12:40:08.807501Z","submitted_at":"2025-05-29T16:11:18Z","title":"LLM Performance for Code Generation on Noisy Tasks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:45:41.067226Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2505.23598"},"observation_digest":"sha256:5eabd2e7b35c22c38a0425b7b6243fafd01435d06c30d9fec3806af25204ddd1","observation_id":"dc4387f9-e218-4af0-a674-4f10df6c376e","resolution":{"observed_at":"2026-08-07T12:45:41.067226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-06T22:30:30.708684Z","title":"Dynabench: Rethinking benchmarking in nlp","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.21521","last_updated":"2025-06-29T18:12:45Z","snapshot_observed_at":"2026-08-07T23:44:24.327568Z","submitted_at":"2025-06-26T17:41:35Z","title":"Potemkin Understanding in Large Language Models","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T22:30:30.708684Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2506.21521"},"observation_digest":"sha256:880c835307e66321b76a55a39bdc43f12c2e9024ed3cbb3c33299d3a8cc50e75","observation_id":"6c072be9-1d61-4f5e-97e5-5ac8b700aab8","resolution":{"observed_at":"2026-08-06T22:30:30.708684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-06T17:08:15.386951Z","title":"Kiela, M","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.13383","last_updated":"2025-07-15T21:02:35Z","snapshot_observed_at":"2026-08-06T17:00:41.170928Z","submitted_at":"2025-07-15T21:02:35Z","title":"Whose View of Safety? A Deep DIVE Dataset for Pluralistic Alignment of Text-to-Image Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T17:08:15.386951Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2507.13383"},"observation_digest":"sha256:27d85da442524d54637efdc3a084ef54138d678f6b5590b5e67775a59c57e810","observation_id":"011ad500-ec58-4f4c-8ec8-45ea96369d38","resolution":{"observed_at":"2026-08-06T17:08:15.386951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-06T13:05:37.910504Z","title":"Dynabench: Rethinking benchmarking in nlp","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.21206","last_updated":"2025-07-28T17:58:12Z","snapshot_observed_at":"2026-08-08T10:48:52.442522Z","submitted_at":"2025-07-28T17:58:12Z","title":"Agentic Web: Weaving the Next Web with AI Agents","version":1},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-08-06T13:05:37.910504Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2507.21206"},"observation_digest":"sha256:6ed11d1a664289187f487bac5f6a656499fa496d20de2626bba0bd8a7076f89c","observation_id":"2282e2fc-22ea-45e8-86ae-724b13ad9049","resolution":{"observed_at":"2026-08-06T13:05:37.910504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2507.22359","last_updated":"2026-04-14T11:47:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-30T03:50:46Z","title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","version":4},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-19T03:17:06.457421Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2507.22359"},"observation_digest":"sha256:ffb5174fc4522ef186f86b5f1339a52a6bb366f62acd5376db906ae5e295a762","observation_id":"6cf568cf-ac71-422f-8d4f-e25c6af4ebed","resolution":{"observed_at":"2026-05-19T03:22:01.364886Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-05T15:43:59.083960Z","title":"Dynabench: Rethinking benchmarking in nlp.arXiv preprint arXiv:2104.14337, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00085","last_updated":"2025-08-27T07:08:44Z","snapshot_observed_at":"2026-08-07T12:15:31.193814Z","submitted_at":"2025-08-27T07:08:44Z","title":"Private, Verifiable, and Auditable AI Systems","version":1},"reference_index":153,"source":"pdf_text","source_observed_at":"2026-08-05T15:43:59.083960Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2509.00085"},"observation_digest":"sha256:fd6c70f9072f8512e55c09d88b7c4189243df43e0e4ca6750f4ffa2bac45092c","observation_id":"b6ad4775-1b20-42e1-9058-1c72ab173879","resolution":{"observed_at":"2026-08-05T15:43:59.083960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-04T19:48:22.922729Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.09055","last_updated":"2025-09-10T23:22:59Z","snapshot_observed_at":"2026-08-07T05:43:24.072429Z","submitted_at":"2025-09-10T23:22:59Z","title":"Improving LLM Safety and Helpfulness using SFT and DPO: A Study on OPT-350M","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T19:48:22.922729Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2509.09055"},"observation_digest":"sha256:10427547e92f2cd630f6665a9920f80e3f7d23881d84d6cdd97401f688dfb276","observation_id":"91451675-098c-4626-8d54-ab661566828f","resolution":{"observed_at":"2026-08-04T19:48:22.922729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2510.09275","last_updated":"2026-04-20T09:54:16Z","snapshot_observed_at":"2026-08-02T12:14:24.192489Z","submitted_at":"2025-10-10T11:19:04Z","title":"Inflated Excellence or True Performance? Rethinking Medical Diagnostic Benchmarks with Dynamic Evaluation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T08:12:02.449352Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2510.09275"},"observation_digest":"sha256:2d5df60d19aca47551b566a3ed1d2d50eed0ee379add587790db2e09597d849b","observation_id":"0c5adf30-311c-4f75-b382-796f0467f70f","resolution":{"observed_at":"2026-05-18T08:12:29.823956Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-03T12:21:38.373016Z","title":"Preprint, arXiv:2104.14337","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.03471","last_updated":"2026-05-26T12:28:30Z","snapshot_observed_at":"2026-08-03T12:21:36.165758Z","submitted_at":"2026-01-06T23:49:10Z","title":"EpiQAL: Benchmarking Large Language Models in Epidemiological Question Answering and Reasoning","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T12:21:38.373016Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2601.03471"},"observation_digest":"sha256:30af164bcb62eaf79fe3b399e71c874e19940769b7a57a37fdebdb4d2e37e01b","observation_id":"12d933b8-0880-4d12-86f6-c66322eed828","resolution":{"observed_at":"2026-08-03T12:21:38.373016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2603.14987","last_updated":"2026-05-21T06:24:06Z","snapshot_observed_at":"2026-08-03T21:30:30.993382Z","submitted_at":"2026-03-16T08:51:33Z","title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-22T10:19:56.003219Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2603.14987"},"observation_digest":"sha256:aeb736a2cf932f6c0daf129872581d82fa54ca74ea474080c902d5d6fb20bb22","observation_id":"7fa4cbc4-7498-4013-9463-71b8fa7e0024","resolution":{"observed_at":"2026-05-22T10:21:23.268582Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2604.05226","last_updated":"2026-04-06T22:42:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T22:42:05Z","title":"RoboPlayground: Democratizing Robotic Evaluation through Structured Physical Domains","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T18:46:08.897540Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2604.05226"},"observation_digest":"sha256:dc60ef6f4e18b55a51346cb770a50d2c01580dfd84b32e2b6340196d6a987eb3","observation_id":"142def31-9698-490a-a6c7-a17862a62603","resolution":{"observed_at":"2026-05-11T00:00:52.133066Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2604.07593","last_updated":"2026-06-17T23:58:56Z","snapshot_observed_at":"2026-07-13T08:26:43.817939Z","submitted_at":"2026-04-08T20:51:00Z","title":"Too long; didn't solve","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T17:29:25.830962Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2604.07593"},"observation_digest":"sha256:c114c58a852d131fd54421ad7f5f7dcb0ee462013ed44942d32b9f9ebed03f42","observation_id":"04341ae4-8797-4e6b-9c8b-15d01ebd9d6c","resolution":{"observed_at":"2026-05-11T06:41:42.876444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-13T08:26:49.097626Z","title":"ArXiv:2104.14337","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.07593","last_updated":"2026-06-17T23:58:56Z","snapshot_observed_at":"2026-07-13T08:26:43.817939Z","submitted_at":"2026-04-08T20:51:00Z","title":"Too long; didn't solve","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-13T08:26:49.097626Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2604.07593"},"observation_digest":"sha256:037e58dc85dcde8f1d53d9eca06eee810c5f75911de363fe1dc627e31d87c650","observation_id":"1382ce5a-86b0-4d76-9a13-c56ed3d8f611","resolution":{"observed_at":"2026-07-13T08:26:49.097626Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2604.16742","last_updated":"2026-08-05T20:20:22Z","snapshot_observed_at":"2026-08-08T15:15:24.028222Z","submitted_at":"2026-04-17T23:18:28Z","title":"CT Open: An Open-Access, Uncontaminated, Live Platform for the Open Challenge of Clinical Trial Outcome Prediction","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T08:02:50.603020Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2604.16742"},"observation_digest":"sha256:384f58ea42a15e1aa807407e3f52f5600c4f2932bf46335c5c1a0eb20abb1c44","observation_id":"00e9179d-a40f-4902-a1e7-dfa5a8904107","resolution":{"observed_at":"2026-05-10T09:18:32.232510Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2604.17842","last_updated":"2026-04-20T05:51:50Z","snapshot_observed_at":"2026-08-04T02:34:16.438625Z","submitted_at":"2026-04-20T05:51:50Z","title":"QuickScope: Certifying Hard Questions in Dynamic LLM Benchmarks","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-10T04:27:11.735657Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2604.17842"},"observation_digest":"sha256:05f6bfe7a6756b255810de3d212c959e32c404ab412e077bff55ca6609e1a3c0","observation_id":"8cbb151b-15f8-4fd4-999b-69396e71f8ff","resolution":{"observed_at":"2026-05-11T11:56:29.611935Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2605.00907","last_updated":"2026-04-29T04:29:48Z","snapshot_observed_at":"2026-07-06T23:14:15.780028Z","submitted_at":"2026-04-29T04:29:48Z","title":"TRIP-Evaluate: An Open Multimodal Benchmark for Evaluating Large Models in Transportation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-09T20:13:06.376037Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2605.00907"},"observation_digest":"sha256:31d442fd8bde0f63b1aca96ea41d2d554fabc48b9e90f85f9ce2ea0944289011","observation_id":"b8ad867c-3b0a-4c9f-b96a-8b3ae646d7ba","resolution":{"observed_at":"2026-05-09T20:17:04.499790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2605.02930","last_updated":"2026-04-27T18:07:58Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-27T18:07:58Z","title":"Analysis and Explainability of LLMs Via Evolutionary Methods","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-09T20:37:40.932811Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2605.02930"},"observation_digest":"sha256:0be1ddbb4120d8ae7a3e4db7a3cc53632b43bccef0c28333752164407713c266","observation_id":"8c63ef9a-691e-4399-b0ed-37d88cdde605","resolution":{"observed_at":"2026-05-11T15:11:05.214722Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2605.04312","last_updated":"2026-05-05T21:24:58Z","snapshot_observed_at":"2026-07-06T23:17:04.200299Z","submitted_at":"2026-05-05T21:24:58Z","title":"Agent Island: A Saturation- and Contamination-Resistant Benchmark from Multiagent Games","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-08T17:06:32.814188Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2605.04312"},"observation_digest":"sha256:1cd836d6e08831565c2d5fdec6fe00c94196c104f35a4824c713d4a88ef20ec1","observation_id":"10663faa-a7ed-4d7c-b0e8-5359b4e29507","resolution":{"observed_at":"2026-05-11T17:46:17.250381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2605.10639","last_updated":"2026-05-11T14:27:39Z","snapshot_observed_at":"2026-08-02T13:54:55.570407Z","submitted_at":"2026-05-11T14:27:39Z","title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T05:28:45.453455Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2605.10639"},"observation_digest":"sha256:d375b58b8f5e0d6f1a492ee61afc96cfe3a0c717e1983a87838a53d516209fe5","observation_id":"3f039523-3804-412f-a2bc-39b04018becd","resolution":{"observed_at":"2026-05-12T05:31:23.769739Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2605.17829","last_updated":"2026-05-18T04:03:18Z","snapshot_observed_at":"2026-08-05T06:31:55.615676Z","submitted_at":"2026-05-18T04:03:18Z","title":"Interactive Evaluation Requires a Design Science","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-20T10:55:08.135630Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2605.17829"},"observation_digest":"sha256:303df100e82dd2943514ba3240f265a76822d6f344061df45a313acab2a0ddb8","observation_id":"f0e8d7ea-342f-4d28-bfcd-1ae36ec07607","resolution":{"observed_at":"2026-05-20T10:58:14.040649Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2605.20520","last_updated":"2026-05-19T21:42:32Z","snapshot_observed_at":"2026-08-03T02:30:24.579386Z","submitted_at":"2026-05-19T21:42:32Z","title":"Open-World Evaluations for Measuring Frontier AI Capabilities","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T06:38:51.427985Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2605.20520"},"observation_digest":"sha256:012a77fe0c9ba18c024fd5e870fb41f01315a8f05ddb04b0c56a1207c3f6a3dc","observation_id":"e6026c7b-c1ff-4fa7-b33c-fd7ec7ec5ce6","resolution":{"observed_at":"2026-05-21T06:39:43.734758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2606.03650","last_updated":"2026-06-04T10:01:47Z","snapshot_observed_at":"2026-08-06T12:17:15.155456Z","submitted_at":"2026-06-02T13:41:43Z","title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T10:46:24.554332Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2606.03650"},"observation_digest":"sha256:ee9424d30eaa03211dd6e80208eb0653c3ce982905bfcec3745b2f212ab642c8","observation_id":"9137d3af-0e49-425f-865c-49b6bf62c79c","resolution":{"observed_at":"2026-07-02T02:36:27.468788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":"2104.14337","doi":"10.48550/arxiv.2104.14337","metadata_source":"arxiv_reference","pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Dynabench: Rethinking benchmarking in NLP","venue":null,"work_id":"8793a803-9ad7-4fc6-8e3d-11757cca111c","year":2021},"citing_paper":{"arxiv_id":"2607.01740","last_updated":"2026-07-02T05:52:32Z","snapshot_observed_at":"2026-07-07T00:07:19.026006Z","submitted_at":"2026-07-02T05:52:32Z","title":"Meta-Benchmarks for Financial-Services LLM Evaluation","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-03T14:08:20.432931Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2607.01740"},"observation_digest":"sha256:88bc798457c788cb4c7400961a8cf68f901fc6dd99e3858416d1b1862d4f1112","observation_id":"3610d9eb-e494-4f55-87ed-ae935d5d4b59","resolution":{"observed_at":"2026-07-03T14:18:22.686773Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-03T00:24:07.971141Z","title":"emnlp-main.306/","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.28801","last_updated":"2026-07-30T19:51:55Z","snapshot_observed_at":"2026-08-06T19:32:03.142991Z","submitted_at":"2026-07-30T19:51:55Z","title":"Benchmarks Are Not Monolithic: Sample-Level Auditing and Orchestration for LLM Evaluation","version":1},"reference_index":306,"source":"pdf_text","source_observed_at":"2026-08-03T00:24:07.971141Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2607.28801"},"observation_digest":"sha256:4540768c78c8d779c181e19f0149aebd1e0e482862d35d1cc82ac03cba68f575","observation_id":"d18b0e1a-4d83-4284-a968-2162598b044d","resolution":{"observed_at":"2026-08-03T00:24:07.971141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.14337","snapshot_observed_at":"2026-08-07T04:45:33.194776Z","title":"arXiv preprint arXiv:2104.14337 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.06352","last_updated":"2026-08-06T17:53:18Z","snapshot_observed_at":"2026-08-08T15:20:27.879148Z","submitted_at":"2026-08-06T17:53:18Z","title":"CalibForge: Adversarial Solver Calibration for Scaling Learnable Terminal Tasks","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T04:45:33.194776Z"},"links":{"cited_paper":"/paper/2104.14337","citing_paper":"/paper/2608.06352"},"observation_digest":"sha256:2c20401e1d59d2185bca8139c695a3f77a4b0ad52c5a613e290c746a772a8d98","observation_id":"b805c310-4960-4f48-ab76-5432e72d93d6","resolution":{"observed_at":"2026-08-07T04:45:33.194776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2104.14337/citation-record","integrity":"/paper/2104.14337/integrity","json":"/paper/2104.14337/citation-record.json","paper":"/paper/2104.14337"},"outbound":[],"paper":{"arxiv_id":"2104.14337","last_updated":"2021-04-07T17:49:17Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T23:24:34.241550Z","submitted_at":"2021-04-07T17:49:17Z","title":"Dynabench: Rethinking Benchmarking in NLP"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2104.14337."}