{"as_of":"2026-08-09T18:53:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:854e4dc10d54c4b074f9f8baa8b9dd2eec5a56da59356fc8ebee571d5cc5479a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":25,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T04:54:50.853489Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T02:49:24.847192Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2310.08419","last_updated":"2024-07-18T18:24:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-12T15:38:28Z","title":"Jailbreaking Black Box Large Language Models in Twenty Queries","version":4},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-12T09:48:31.721745Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2310.08419"},"observation_digest":"sha256:9a3da10e6e6ed3972068b3e9fede60e0bd37eb5a988a7a892bbd4d2d89dc6f6c","observation_id":"5caf5a5a-1988-475f-8864-1b178e1e3610","resolution":{"observed_at":"2026-05-12T09:48:33.255295Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-08T04:54:50.853489Z","title":"T., Wu, T., Guestrin, C., and Singh, S","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2502.08512","last_updated":"2025-08-14T07:15:48Z","snapshot_observed_at":"2026-08-09T13:56:26.887562Z","submitted_at":"2025-02-12T15:46:34Z","title":"Measuring Diversity in Synthetic Datasets","version":3},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-08T04:54:50.853489Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2502.08512"},"observation_digest":"sha256:69d9ca93a394d5f43f3b0942dcc5252a76b89d4105cfff3d1fbd7ab9fd6924cc","observation_id":"c6fbf12b-d000-483f-9986-45df04151659","resolution":{"observed_at":"2026-08-08T04:54:50.853489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-07T12:14:10.675956Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.00134","last_updated":"2025-05-30T18:11:33Z","snapshot_observed_at":"2026-08-09T02:52:42.303020Z","submitted_at":"2025-05-30T18:11:33Z","title":"Spurious Correlations and Beyond: Understanding and Mitigating Shortcut Learning in SDOH Extraction with Large Language Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T12:14:10.675956Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2506.00134"},"observation_digest":"sha256:60dd80f6213e8fe12c468ffe27c6900604ae5e251ddd4c0070eeb79ffaabb9e7","observation_id":"f1b3c440-7823-4e4c-b899-9957d5f700d2","resolution":{"observed_at":"2026-08-07T12:14:10.675956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-07T11:14:42.214448Z","title":"Beyond accuracy: Behavioral testing of nlp models with check- list","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.03024","last_updated":"2025-06-03T16:00:30Z","snapshot_observed_at":"2026-08-09T02:51:34.085222Z","submitted_at":"2025-06-03T16:00:30Z","title":"GenFair: Systematic Test Generation for Fairness Fault Detection in Large Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:14:42.214448Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2506.03024"},"observation_digest":"sha256:306f81cdd8669bd6d5e55d884f568689a61546fbeb88069150905f21fd81a4f4","observation_id":"44185dd5-4404-45c2-b8e8-5e3dcf2f7052","resolution":{"observed_at":"2026-08-07T11:14:42.214448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-06T22:30:31.993054Z","title":"T., Wu, T., Guestrin, C., and Singh, S","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2506.21521","last_updated":"2025-06-29T18:12:45Z","snapshot_observed_at":"2026-08-09T04:44:37.252637Z","submitted_at":"2025-06-26T17:41:35Z","title":"Potemkin Understanding in Large Language Models","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T22:30:31.993054Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2506.21521"},"observation_digest":"sha256:e868ad56c30c722ca299c2694d2429fe7228d92c180521b7078b83f0fc8e3b56","observation_id":"1bbb4541-0b04-4f46-83a6-dcf80abbd733","resolution":{"observed_at":"2026-08-06T22:30:31.993054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-06T19:43:29.244466Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.05307","last_updated":"2025-07-07T09:11:16Z","snapshot_observed_at":"2026-08-09T15:39:54.255301Z","submitted_at":"2025-07-07T09:11:16Z","title":"ASSURE: Metamorphic Testing for AI-powered Browser Extensions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T19:43:29.244466Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2507.05307"},"observation_digest":"sha256:2608db9f3f1c05ce3ff89bc9ba88ee2cb3ae9dd7d91bcbf6641c3aa72779f942","observation_id":"ff9351b1-e60b-4ae7-a2cd-11fd9d455bf6","resolution":{"observed_at":"2026-08-06T19:43:29.244466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-06T13:05:39.875211Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.21206","last_updated":"2025-07-28T17:58:12Z","snapshot_observed_at":"2026-08-08T10:48:52.442522Z","submitted_at":"2025-07-28T17:58:12Z","title":"Agentic Web: Weaving the Next Web with AI Agents","version":1},"reference_index":186,"source":"arxiv_source","source_observed_at":"2026-08-06T13:05:39.875211Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2507.21206"},"observation_digest":"sha256:2307f20ebeab67c3f8c73282bf2be26555dbcc69386c3085774e5df915d6a8c9","observation_id":"d9db4d8b-1f4d-4a14-a274-8c4700c34de4","resolution":{"observed_at":"2026-08-06T13:05:39.875211Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-05T22:50:43.780478Z","title":null,"venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2508.06426","last_updated":"2025-08-08T16:14:01Z","snapshot_observed_at":"2026-08-07T03:09:20.958694Z","submitted_at":"2025-08-08T16:14:01Z","title":"Shortcut Learning in Generalist Robot Policies: The Role of Dataset Diversity and Fragmentation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T22:50:43.780478Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2508.06426"},"observation_digest":"sha256:34364e56e78e92ca1bcb7373a1e4a919964028e4c941f3b194bf7c6efeeda7c8","observation_id":"ef847eb1-bc82-4595-8096-466b8c60d63f","resolution":{"observed_at":"2026-08-05T22:50:43.780478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-05T22:46:44.391762Z","title":null,"venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2508.06429","last_updated":"2025-08-08T16:16:43Z","snapshot_observed_at":"2026-08-09T06:56:02.033235Z","submitted_at":"2025-08-08T16:16:43Z","title":"SPARSE Data, Rich Results: Few-Shot Semi-Supervised Learning via Class-Conditioned Image Translation","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T22:46:44.391762Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2508.06429"},"observation_digest":"sha256:1cdcb78a60f57432250279d89b6f88e5d13441c716d04f6bf338958aff81f188","observation_id":"997716ef-ddec-459c-8def-2bc06e2f7fb0","resolution":{"observed_at":"2026-08-05T22:46:44.391762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2509.11206","last_updated":"2026-04-20T05:43:30Z","snapshot_observed_at":"2026-08-07T05:35:56.514953Z","submitted_at":"2025-09-14T10:24:13Z","title":"Evalet: Evaluating Large Language Models through Functional Fragmentation","version":4},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-18T16:57:25.259866Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2509.11206"},"observation_digest":"sha256:e72801774566f656c0affcb8e1046fefd3119b4657e1e8e60e92bd7e195d937c","observation_id":"f4c8d66b-1d0a-43ca-ab53-e87408ad5479","resolution":{"observed_at":"2026-05-18T17:01:39.982194Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2511.01458","last_updated":"2026-04-23T14:25:24Z","snapshot_observed_at":"2026-08-01T08:51:46.134196Z","submitted_at":"2025-11-03T11:18:21Z","title":"When to Trust the Answer: Question-Aligned Semantic Nearest Neighbor Entropy for Safer Surgical VQA","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T01:21:26.854636Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2511.01458"},"observation_digest":"sha256:6de41b09a4d1d9bc463131745da89f73b725be2eb9cfd02e6f5f00c38b728c6a","observation_id":"8c9f4954-7740-491d-b601-f9c219ce2473","resolution":{"observed_at":"2026-05-18T01:22:15.719282Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2601.16175","last_updated":"2026-02-05T18:03:03Z","snapshot_observed_at":"2026-08-01T10:41:29.420510Z","submitted_at":"2026-01-22T18:24:00Z","title":"Learning to Discover at Test Time","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-16T05:16:04.001700Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2601.16175"},"observation_digest":"sha256:4c152e0f12cb736fa26a951f823dd8e98dd20fc4b4a76b9e0b680e59aa0a6e35","observation_id":"26e056f2-14f0-433a-a8d8-32a364ff2adf","resolution":{"observed_at":"2026-05-16T05:16:04.194337Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2604.16421","last_updated":"2026-04-03T11:36:49Z","snapshot_observed_at":"2026-07-06T23:03:43.854612Z","submitted_at":"2026-04-03T11:36:49Z","title":"Measuring Representation Robustness in Large Language Models for Geometry","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T19:35:32.531660Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2604.16421"},"observation_digest":"sha256:2baae3b8040406493b414a755b20ca26cf4c194930d41b87f54de61b836e782d","observation_id":"97b09954-a9ce-4b3f-864e-8c7ddecbcdf8","resolution":{"observed_at":"2026-05-13T19:38:10.520784Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2605.00382","last_updated":"2026-05-05T08:05:07Z","snapshot_observed_at":"2026-07-06T23:13:47.082550Z","submitted_at":"2026-05-01T04:06:02Z","title":"Social Bias in LLM-Generated Code: Benchmark and Mitigation","version":3},"reference_index":153,"source":"arxiv_source","source_observed_at":"2026-05-09T19:34:51.433422Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2605.00382"},"observation_digest":"sha256:0cf451d091615c0b7b5195f200772cae79b3488c3fa54749b18243c97d760ce1","observation_id":"38f3e839-bca3-4ff1-a7b5-be4ac92820fc","resolution":{"observed_at":"2026-05-11T15:36:08.937852Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2605.13625","last_updated":"2026-05-13T14:52:40Z","snapshot_observed_at":"2026-07-06T23:25:11.623026Z","submitted_at":"2026-05-13T14:52:40Z","title":"How to Interpret Agent Behavior","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-14T18:23:25.269217Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2605.13625"},"observation_digest":"sha256:e6f9281ae4d6779c041745bf5432221cfd1fe22f70f1a40d4b9a5dc027ebc1a8","observation_id":"633d3a0d-5872-477f-a314-cacd2f379e40","resolution":{"observed_at":"2026-05-14T18:27:35.960649Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2605.17829","last_updated":"2026-05-18T04:03:18Z","snapshot_observed_at":"2026-08-05T06:31:55.615676Z","submitted_at":"2026-05-18T04:03:18Z","title":"Interactive Evaluation Requires a Design Science","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-20T10:55:08.135630Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2605.17829"},"observation_digest":"sha256:5fcdc5fb8cf01e204171780952ad614f822f6b452fb6525a28d33a2a2d64f6ab","observation_id":"93a4eb04-fcab-4d4a-8a79-1ffdac38e016","resolution":{"observed_at":"2026-05-20T10:58:14.003198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2606.00540","last_updated":"2026-05-30T05:14:53Z","snapshot_observed_at":"2026-08-01T17:48:34.431096Z","submitted_at":"2026-05-30T05:14:53Z","title":"Trustworthy Recommendation in the Era of Large Language Models: Opportunities and Challenges","version":1},"reference_index":195,"source":"pdf_text","source_observed_at":"2026-06-28T18:41:06.636352Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2606.00540"},"observation_digest":"sha256:bec0202d717b3b1045c4717ee5da67830b3a612753e2d7d71f95cab06f94d6ea","observation_id":"f9bb0636-5f55-41d3-b279-05e048ec2fce","resolution":{"observed_at":"2026-06-28T18:42:29.189519Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2606.04661","last_updated":"2026-06-03T09:40:03Z","snapshot_observed_at":"2026-07-06T23:44:42.700120Z","submitted_at":"2026-06-03T09:40:03Z","title":"CRAFT: Cost-aware Refinement And Front-aware Tuning of Prompts","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-28T06:39:17.268337Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2606.04661"},"observation_digest":"sha256:96d361a1801f2865b3b882d4cb3c9a2e31fada07258ef7adf5e71702cdf95eeb","observation_id":"c6b290ae-5898-429c-baf2-7260f431fd4e","resolution":{"observed_at":"2026-07-02T07:46:46.394764Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2606.09700","last_updated":"2026-06-19T20:34:30Z","snapshot_observed_at":"2026-07-06T23:48:58.206424Z","submitted_at":"2026-06-08T16:21:34Z","title":"What the Eyes See, the LLMs Miss: Exploiting Human Perception for Adversarial Text Attacks","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T16:18:01.850874Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2606.09700"},"observation_digest":"sha256:38648e01a721449a53945ab3599594d40158b193d4da789f78a3f098c1cda4c6","observation_id":"1d509cc0-ac8e-4408-8151-33cc6236c888","resolution":{"observed_at":"2026-07-03T01:47:31.706084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2606.12924","last_updated":"2026-06-11T05:27:42Z","snapshot_observed_at":"2026-07-06T23:51:45.306189Z","submitted_at":"2026-06-11T05:27:42Z","title":"Iterating Toward Better Search: A Two-Agent Simulation Framework for Evaluating Agentic Search Architectures in E-Commerce","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-27T07:05:31.737761Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2606.12924"},"observation_digest":"sha256:4293c3514e1546ba98fa41e34963fd6ed4f433a871eb06928cf516153736a847","observation_id":"1238b618-2488-4f13-bd1e-02eb26ae55dd","resolution":{"observed_at":"2026-07-03T14:18:23.057855Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2606.19057","last_updated":"2026-06-17T13:26:04Z","snapshot_observed_at":"2026-08-07T01:21:32.666151Z","submitted_at":"2026-06-17T13:26:04Z","title":"Quantifying and Auditing LLM Evaluation via Positive--Unlabeled Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-26T19:04:45.062426Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2606.19057"},"observation_digest":"sha256:ebe85ecabf6a6ee20a3032748592a82359c0c62d61d58203ce016eb733113e40","observation_id":"142e94c2-a96f-4559-95d7-a7ec100e14a1","resolution":{"observed_at":"2026-07-04T02:49:24.849283Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":"2005.04118","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-07-04T02:49:24.847192Z","title":"Beyond accuracy: Behavioral testing of nlp models with checklist.ArXiv, abs/2005.04118, 2020.https://api.semanticscholar.org/CorpusID:218551201","venue":null,"work_id":"e147b285-8145-4593-a795-c45ca23c4626","year":2005},"citing_paper":{"arxiv_id":"2606.28360","last_updated":"2026-06-11T20:56:10Z","snapshot_observed_at":"2026-08-03T05:28:58.170719Z","submitted_at":"2026-06-11T20:56:10Z","title":"Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-30T11:13:23.740219Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2606.28360"},"observation_digest":"sha256:01d72a2663596a15b4381841d7ec3e7e997239a95abefec0f1fcbf165cde42be","observation_id":"2e78ebc0-63ab-4619-8c6f-7b5caf2a74da","resolution":{"observed_at":"2026-06-30T11:14:37.507549Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-01T15:54:00.365263Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.18155","last_updated":"2026-07-20T16:53:05Z","snapshot_observed_at":"2026-08-09T05:12:17.754119Z","submitted_at":"2026-07-20T16:53:05Z","title":"Testing Retrieval-Augmented Generation Systems with Chunk Coverage","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T15:54:00.365263Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2607.18155"},"observation_digest":"sha256:e3196fd45bdc9cc39bf1de3ec90291874d9e1cd4de7fb5a3c4120628d31fae95","observation_id":"7c711e76-babd-4a98-80a1-2ccccfe33175","resolution":{"observed_at":"2026-08-01T15:54:00.365263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-02T07:40:17.900025Z","title":"arXiv preprint arXiv:2005.04118 , year=","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.20532","last_updated":"2026-07-10T13:07:12Z","snapshot_observed_at":"2026-08-08T11:53:10.019905Z","submitted_at":"2026-07-10T13:07:12Z","title":"Position: Stop Reactively Patching Your Model Every Time and Start Proactive Test-Driven AI Development","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-02T07:40:17.900025Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2607.20532"},"observation_digest":"sha256:a59ec6f439851681679a3ccf6ecc4b986641615a8ab1f991c5ddee0d8da26eaf","observation_id":"dd85f4fe-5239-4949-89ba-8c79e8ae304e","resolution":{"observed_at":"2026-08-02T07:40:17.900025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04118","snapshot_observed_at":"2026-08-02T13:46:09.525133Z","title":null,"venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.22554","last_updated":"2026-05-18T16:45:13Z","snapshot_observed_at":"2026-08-08T05:06:39.163957Z","submitted_at":"2026-05-18T16:45:13Z","title":"Same Question, Different Answers: Evaluating LLM Reliability Beyond Accuracy","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T13:46:09.525133Z"},"links":{"cited_paper":"/paper/2005.04118","citing_paper":"/paper/2607.22554"},"observation_digest":"sha256:90ef6b29193e77692a1791574eae9bf0c03b4b1b24d8152f77c858bf12d6fa7a","observation_id":"48282414-8bb1-4cee-aeb7-6aafdb4a4c1f","resolution":{"observed_at":"2026-08-02T13:46:09.525133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2005.04118/citation-record","integrity":"/paper/2005.04118/integrity","json":"/paper/2005.04118/citation-record.json","paper":"/paper/2005.04118"},"outbound":[],"paper":{"arxiv_id":"2005.04118","last_updated":"2020-05-08T15:48:31Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T02:47:57.041922Z","submitted_at":"2020-05-08T15:48:31Z","title":"Beyond Accuracy: Behavioral Testing of NLP models with CheckList"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 25 inbound Pith citation observations for arXiv:2005.04118."}