{"as_of":"2026-08-14T14:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6001ca078fde0eba0e2c3194b87839bc2b1efc305c3cec4e820fa00348a82865","coverage":[{"denominator":77,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":77,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T04:15:50.453118Z","state":"measured"},{"denominator":77,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":77,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-14T06:32:32.682623+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.09925/citation-record","integrity":"/paper/2608.09925/integrity","json":"/paper/2608.09925/citation-record.json","paper":"/paper/2608.09925"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-11T04:15:50.013344Z","title":"arXiv preprint arXiv:2507.06261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.013344Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:dc6194a3601b8c1e3c37960031d5150ff9ce60f532df5adf06e0429b178b9ddb","observation_id":"65bda534-6390-4a18-936c-c885c4249b49","resolution":{"observed_at":"2026-08-11T04:15:50.013344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T04:15:50.020264Z","title":"arXiv preprint arXiv:2303.08774 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.020264Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:1d7fc9d21a9adab3e3cd63e055ea6261826aa137890076168133f021af67d09e","observation_id":"f2ff29d1-b08e-45ba-9da7-4ef0ce1eb598","resolution":{"observed_at":"2026-08-11T04:15:50.020264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-11T04:15:50.026324Z","title":"arXiv preprint arXiv:2302.13971 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.026324Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:23e5846614ca5d58ca3a4cb0262815f4e10a7ca338274ddeb47aa8085e2ae85a","observation_id":"a43fa297-80fb-423e-aa19-1fc4cbd1bb1b","resolution":{"observed_at":"2026-08-11T04:15:50.026324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.031881Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.031881Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:c19e4491e234bd11b1eb7d62106e1c4093428bb37ea0fcaee524b1ed440185b1","observation_id":"1feaeb13-ec9b-43fb-8f14-a2d20ffa772c","resolution":{"observed_at":"2026-08-11T04:15:50.031881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.037460Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.037460Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:5531cc53f62a7b1239e94e38e09c72d022f1c1f02e092c75e6bb52a41951708d","observation_id":"27dca7fa-e952-4fbf-a1d5-2a6ee71605ec","resolution":{"observed_at":"2026-08-11T04:15:50.037460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.125637Z","title":"Telecommunications policy , volume=","venue":null,"work_id":"709dabf4-20f5-4193-9019-65b5c4d07fd0","year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.046129Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:b891d7678e9270d8be64ec4cbc5d126b7fd37fbb002c2cbba1b05ce6f8f56f8b","observation_id":"59f90afa-1d14-434e-b1be-53896cf2b46e","resolution":{"observed_at":"2026-08-11T04:16:03.131532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.107656Z","title":"Government information quarterly , volume=","venue":null,"work_id":"6be18835-d4e1-40f9-9161-38b9c8a2069d","year":2022},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.057018Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:e0e13a32d63516c73404fe470cfbaa79c19f18f4a490fc4148ec62fc04343868","observation_id":"f3d8d368-4e0b-4424-973a-bc438debc1ca","resolution":{"observed_at":"2026-08-11T04:16:03.112621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.091331Z","title":"European Journal of Social Security , volume=","venue":null,"work_id":"6a40adb9-1ed2-45f2-89ad-ce2435187343","year":2021},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.062683Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:85b5af598006825ef9672022438b38b69a1b72a70dee66833e949269ba9f4f7a","observation_id":"509f096f-ebf8-4f9a-8c85-27441266277f","resolution":{"observed_at":"2026-08-11T04:16:03.096441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.068967Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.068967Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:760e93d8eab25ef5bf659f66df19c174857d483049db4de5f67c903f6ca4a22c","observation_id":"c67244c6-792e-4b89-b4e7-957c90311c70","resolution":{"observed_at":"2026-08-11T04:15:50.068967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.19799","last_updated":"2024-11-29T16:03:14Z","snapshot_observed_at":"2026-08-14T11:28:04.777011Z","submitted_at":"2024-11-29T16:03:14Z","title":"INCLUDE: Evaluating Multilingual Language Understanding with Regional Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.19799","snapshot_observed_at":"2026-08-11T04:15:50.074768Z","title":"arXiv preprint arXiv:2411.19799 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.074768Z"},"links":{"cited_paper":"/paper/2411.19799","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:cde3be8ef5e98aa206a9aab51dca089945930e64a55607bc68579dfb5b138c9f","observation_id":"7bef12fe-a67b-48ca-85ce-ffcb8c8d04a5","resolution":{"observed_at":"2026-08-11T04:15:50.074768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.063085Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":"d5973957-136a-40b1-a4e7-8f1db8f38e0b","year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.082059Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:38fc41e233c23a492a6b2ddbbecc8a097c3cda422c6683aec0fbf41679bcbbc4","observation_id":"2d3ed7f8-2415-431c-9c42-617aaea06e9a","resolution":{"observed_at":"2026-08-11T04:16:03.068623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.046318Z","title":"Transactions on Machine Learning Research , year=","venue":null,"work_id":"f74c9d33-ac88-4a1e-9d66-85f63743ec85","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.088414Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6e179238b27e0e95837d99051f63e570c9b42127b3025b81e3538ba28a733d1d","observation_id":"3740ec46-eea5-4231-85a7-a72392c512c1","resolution":{"observed_at":"2026-08-11T04:16:03.051480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.029242Z","title":"GEITje: een groot open Nederlands taalmodel , shorttitle =","venue":null,"work_id":"382aff9f-7070-4604-a5da-d4bfb11ce8fa","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.093926Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:13fb06acbc98366124ceffc2aa6b09213f63fbe20997cc275c2c089f91e058da","observation_id":"547bf310-5a75-41ae-9498-e7edd200614c","resolution":{"observed_at":"2026-08-11T04:16:03.034277Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.099046Z","title":"Proceedings of the 57th annual meeting of the association for computational linguistics , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.099046Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:b4bfc4857f845b0e97d8fecb68eead9784b4ca8aa53f67e9717fc7832c4e5076","observation_id":"3ca86f3d-54af-4865-82db-81fe058d544c","resolution":{"observed_at":"2026-08-11T04:15:50.099046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.001708Z","title":"Proceedings of the 2024 ACM conference on fairness, accountability, and transparency , pages=","venue":null,"work_id":"8afebc56-528b-43f4-af71-0179cc995eeb","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.105012Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:72f186865c5f9fc52b361f15490310dde25ace78d882293f0c25064d3056574a","observation_id":"a818207d-db1c-45ae-823e-da6dc2a7992b","resolution":{"observed_at":"2026-08-11T04:16:03.006747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.09700","last_updated":"2019-11-04T20:37:33Z","snapshot_observed_at":"2026-08-13T09:44:21.186377Z","submitted_at":"2019-10-21T23:57:32Z","title":"Quantifying the Carbon Emissions of Machine Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.09700","snapshot_observed_at":"2026-08-11T04:15:50.111292Z","title":"arXiv preprint arXiv:1910.09700 , year=","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.111292Z"},"links":{"cited_paper":"/paper/1910.09700","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2f24cb72158fa23b67fad2d6801fea58c29c3913a7322b68591cfe439da0b64f","observation_id":"9d8872d5-a51e-4184-90c1-d75aeee6e346","resolution":{"observed_at":"2026-08-11T04:15:50.111292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.09582","last_updated":"2019-12-19T22:59:26Z","snapshot_observed_at":"2026-08-12T20:25:16.063548Z","submitted_at":"2019-12-19T22:59:26Z","title":"BERTje: A Dutch BERT Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.09582","snapshot_observed_at":"2026-08-11T04:15:50.117536Z","title":"arXiv preprint arXiv:1912.09582 , year=","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.117536Z"},"links":{"cited_paper":"/paper/1912.09582","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:137af55df7cbc8e3a9012b564d40e6c070bead910b276804b4c6e5ae1cb1489f","observation_id":"f9bee712-8313-4257-8894-5256bb14317e","resolution":{"observed_at":"2026-08-11T04:15:50.117536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.984457Z","title":"Findings of the association for computational linguistics: EMNLP 2020 , pages=","venue":null,"work_id":"21f33ebc-4fa8-4c02-b67e-a40f26e101c7","year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.122939Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:10c9f663f1bd28e990e2536862f27ac449f7ffe61af3e49426553ce70f0079b1","observation_id":"32263682-70fe-4c63-8403-e35e26ade5a1","resolution":{"observed_at":"2026-08-11T04:16:02.990092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13469","last_updated":"2025-01-11T10:20:26Z","snapshot_observed_at":"2026-08-14T10:43:53.830861Z","submitted_at":"2024-06-19T11:50:09Z","title":"Encoder vs Decoder: Comparative Analysis of Encoder and Decoder Language Models on Multilingual NLU Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13469","snapshot_observed_at":"2026-08-11T04:15:50.128015Z","title":"arXiv preprint arXiv:2406.13469 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.128015Z"},"links":{"cited_paper":"/paper/2406.13469","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:58ed3887debeeff074df0f9f8c9cc53d3defaa27179d2053ec222a697097282a","observation_id":"dbf5f15b-b203-4a95-8a50-fa5b47e00b3d","resolution":{"observed_at":"2026-08-11T04:15:50.128015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.968153Z","title":null,"venue":null,"work_id":"5881f0de-f050-41c9-919f-50c9136af65f","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.133850Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:9924732c1f654349d00380fd9bb480f9f9e4a98800f77974ed0c1b34f7e5a256","observation_id":"82fbc123-7f99-4980-bf17-de6436fc4ce1","resolution":{"observed_at":"2026-08-11T04:16:02.972604Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.952230Z","title":"2025 , month =","venue":null,"work_id":"51981072-14f5-4e03-a20b-d32f588dec8c","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.138883Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:fa8d10617111c81d42d7e4f52f2ade9f3c35f6d6f81f29c2ab7bbb05947510ee","observation_id":"db03c3ec-db51-4a11-a72e-c0ad93ffdfb0","resolution":{"observed_at":"2026-08-11T04:16:02.957079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.936402Z","title":"2025 , month =","venue":null,"work_id":"3526b611-cf4a-4cab-bfaf-4ee7c279177e","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.144427Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:5bae2bffc1303a3df56cc30ad92820c7c1347a255a8e98d89a0a54cf37d15f8e","observation_id":"20c60e0c-a7fb-47d9-9025-e24bf3f2ac88","resolution":{"observed_at":"2026-08-11T04:16:02.941272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02446","last_updated":"2024-01-27T22:54:52Z","snapshot_observed_at":"2026-08-13T01:25:06.346704Z","submitted_at":"2023-10-03T21:30:56Z","title":"Low-Resource Languages Jailbreak GPT-4","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02446","snapshot_observed_at":"2026-08-11T04:15:50.149734Z","title":"arXiv preprint arXiv:2310.02446 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.149734Z"},"links":{"cited_paper":"/paper/2310.02446","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:948d03ecc5b54f913cabe56e09c6152c93392d76edc3048106d6fd100b73c12b","observation_id":"749fd45f-f6c6-445f-80e0-106a461a5a3b","resolution":{"observed_at":"2026-08-11T04:15:50.149734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.919765Z","title":"Pending submission","venue":null,"work_id":"ef324bb1-5d6d-4215-804c-39a60fc15085","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.155277Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:34331a91ec2a06e9600a6ec183a9b92f7bbb106f18c7181766abf78a52de5d82","observation_id":"2b544e77-1e70-46c8-901d-9253df04b52e","resolution":{"observed_at":"2026-08-11T04:16:02.924779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.903497Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"2a11134e-1385-448a-89b3-f8de2fb262d0","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.161741Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:621034fddeff3db71404d9f011e50b0c17900f84769d8012ecf05dc774031cc1","observation_id":"287dbf20-d41e-4dc2-b085-5002b5ff2516","resolution":{"observed_at":"2026-08-11T04:16:02.908434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-12T23:39:30.739911Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-11T04:15:50.166929Z","title":"arXiv preprint arXiv:2406.13261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.166929Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:a364b72fef29de4824629eb3f723271d366515ef8528bc8a0a8f15001c34c00e","observation_id":"fc86b4ff-f50e-443f-b78b-adfa071959e4","resolution":{"observed_at":"2026-08-11T04:15:50.166929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.01768","last_updated":"2023-01-05T07:13:13Z","snapshot_observed_at":"2026-08-13T13:09:30.284750Z","submitted_at":"2023-01-05T07:13:13Z","title":"The political ideology of conversational AI: Converging evidence on ChatGPT's pro-environmental, left-libertarian orientation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.01768","snapshot_observed_at":"2026-08-11T04:15:50.172952Z","title":"arXiv preprint arXiv:2301.01768 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.172952Z"},"links":{"cited_paper":"/paper/2301.01768","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:0de6811b239dcbb8ad2100bdfbd28e76391b7c5ba6482c7aa67e27a82954bace","observation_id":"2d9a8600-25bd-4edc-8688-0bafe4b0586e","resolution":{"observed_at":"2026-08-11T04:15:50.172952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.178616Z","title":"PloS one , volume=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.178616Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:b9da5360290a8ea985be3b115e4f318345146c8a838a831f9f0f8e97d58df2d8","observation_id":"5f15cbef-9b6d-4f9d-ba47-2041c409fc0c","resolution":{"observed_at":"2026-08-11T04:15:50.178616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.876576Z","title":"Proceedings of the AAAI/ACM Conference on AI, Ethics, and Society , year=","venue":null,"work_id":"3d87cb07-7846-4597-a0aa-0d2678e9eef1","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.184072Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:a1b377a9a555667771f2afdb83815cfa26fbd06b35ebf2c37566baf0db00eebc","observation_id":"224fa6fa-39c4-4536-9b8b-a3b8aff85088","resolution":{"observed_at":"2026-08-11T04:16:02.882505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.190324Z","title":"arXiv preprint arXiv:2601.08785 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.190324Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:dd13b04cfa5e4859e6555475cc99a3ea909c1d7cc70011b48229829febc2f659","observation_id":"22933243-28db-4a31-b839-6d2ef8753e7a","resolution":{"observed_at":"2026-08-11T04:15:50.190324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-11T04:15:50.195330Z","title":"arXiv preprint arXiv:1803.05457 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.195330Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:29abbd2fe194886353d00b18889941dc5dfca9a3d9c4d9d555e7c5847c995829","observation_id":"ea88fa7e-2a71-4bc5-8928-94ea95e64a67","resolution":{"observed_at":"2026-08-11T04:15:50.195330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-14T07:50:45.743715Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-11T04:15:50.201032Z","title":"arXiv preprint arXiv:2402.14992 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.201032Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:7390864a3eb7e14b78ff22b81ecc298795c2faaead175e195bfc2f52a0bb6f71","observation_id":"f6658ab6-13d1-4a6b-9915-7a3c8ce8e323","resolution":{"observed_at":"2026-08-11T04:15:50.201032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.859160Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing: System Demonstrations , pages=","venue":null,"work_id":"8a1675ac-9c66-4d7b-9d28-703fb9c72c3f","year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.207001Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6f003b2c3d40d35719d930013fd6eaf2e3ecfc5866159e43fd8a80be9ca79729","observation_id":"2f9ac80b-b3ab-4b01-ad5f-2bfc61acced6","resolution":{"observed_at":"2026-08-11T04:16:02.864600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.213983Z","title":"Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP) , pages=","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.213983Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:ce89eed8ed8b3213e8037da976c27168dcdae0f8d32f4e71c297630aa7720dd3","observation_id":"6aa152bb-944e-4066-89a8-6c6ba9f70849","resolution":{"observed_at":"2026-08-11T04:15:50.213983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.831811Z","title":"Proceedings of the 15th International Conference on Recent Advances in Natural Language Processing-Natural Language Processing in the Generative AI Era , pages=","venue":null,"work_id":"062f6bde-27d3-449a-a8a6-199fc4f459c2","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.219523Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:7437a06baf4ca142001b9e23ceac1b318a66cdf149624ab71404704433a0d7b6","observation_id":"159dd60e-07d8-4329-8bb7-fd9cddd64f00","resolution":{"observed_at":"2026-08-11T04:16:02.837311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.815033Z","title":null,"venue":null,"work_id":"1ac9ab6e-1f5e-4633-8bc9-6e152c91ebaa","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.224827Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:a6a3d651ac27635d1084965e0dc33923ff9abc821ff949ccb71a13692afd4fc0","observation_id":"9fe2a5e1-15c8-42dd-bd8a-2189dac4fd9c","resolution":{"observed_at":"2026-08-11T04:16:02.820106Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.230578Z","title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 1: long papers) , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.230578Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6422f70ab7a51d3f013d3ce7d500b114400d2397589d1b481954c8a11c79970c","observation_id":"6a7f0154-340c-4fdc-b1a5-0465ddbcd5a4","resolution":{"observed_at":"2026-08-11T04:15:50.230578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.235724Z","title":"Findings of the Association for Computational Linguistics: ACL 2022 , pages=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.235724Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:45453e47ac1e47d882ad29342d7491913d8535c49b64016654fee83cc4c509eb","observation_id":"ae4fd28e-bb08-49ed-a7b3-72da331302df","resolution":{"observed_at":"2026-08-11T04:15:50.235724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-11T04:15:50.241431Z","title":"arXiv preprint arXiv:2009.03300 , year=","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.241431Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:76cf2304aa0d943f953040ff053b0cd5e717dfa8d198fcdd892fde1464171c26","observation_id":"c2601e12-de8c-4528-9db1-449b9b46df53","resolution":{"observed_at":"2026-08-11T04:15:50.241431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07243","last_updated":"2024-07-17T08:49:22Z","snapshot_observed_at":"2026-08-13T01:19:31.948244Z","submitted_at":"2024-06-11T13:23:14Z","title":"MBBQ: A Dataset for Cross-Lingual Comparison of Stereotypes in Generative LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07243","snapshot_observed_at":"2026-08-11T04:15:50.247552Z","title":"arXiv preprint arXiv:2406.07243 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.247552Z"},"links":{"cited_paper":"/paper/2406.07243","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:9b7dd80e9385e7e8c235d05f8f7adcd20e7254c2edff3f756bb59b95fae997b9","observation_id":"04e13413-6b6c-4120-a6b7-2cbf99295609","resolution":{"observed_at":"2026-08-11T04:15:50.247552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.775834Z","title":"Proceedings of the 2023 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":"5e02034a-09af-47bd-828b-04e6ada351ac","year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.254727Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:331a9ddddd13d495c5efa977d7a22c1b17c132e0b2e2afd5fde69dbd79cb2fc2","observation_id":"1f9e392f-045f-49c1-8ec5-a8beab3e9a77","resolution":{"observed_at":"2026-08-11T04:16:02.781074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.758876Z","title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (lrec-coling 2024) , pages=","venue":null,"work_id":"6c4d226a-f84d-4a81-a826-2d429e68b56a","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.261054Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:144e975e0cc5edab2202644f1dae6ea0488d910947f2bfd0996e0c31d3adf032","observation_id":"fe66995b-c666-4f1c-bd04-f87974963e67","resolution":{"observed_at":"2026-08-11T04:16:02.764354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.742505Z","title":"Online:< https://ticclops","venue":null,"work_id":"94ea9b85-c964-4f69-97ff-3ac49258747a","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.267364Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:370f6d51ca9af67f996a9f5833cb62f91c566debcad5eec3a749ece16a0ec630","observation_id":"43498d15-060c-4f3a-b39e-2cf7f9b0c2b9","resolution":{"observed_at":"2026-08-11T04:16:02.747690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.726276Z","title":"CLARIN Annual Conference Proceedings , pages=","venue":null,"work_id":"d235756f-c0c8-4b67-8ebc-b4c9f2430156","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.273694Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:443164e4c23e06b45d7a46fa4ae7caa4ebb2077ef1d5c6cbc7b50cd2cce50f97","observation_id":"98e3bea8-9de1-4ead-870d-bab5db8c0dcf","resolution":{"observed_at":"2026-08-11T04:16:02.731379Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.706956Z","title":"Proceedings of the 2018 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":"93e751a7-e94d-461b-a4a9-773b15e88cea","year":2018},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.279029Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:f70553f9e83d40c7dc5266f994b9e394af518a01c61e489a11c9fa87aff9ad72","observation_id":"ef2f24ce-dd09-4404-be85-34d247c262d6","resolution":{"observed_at":"2026-08-11T04:16:02.712268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.284065Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.284065Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:8d18718c58efc5896c0df9c42e4016e1513690000fe953fdb4b0c2fa5a42f07a","observation_id":"702c8270-c0fb-4578-b5ee-e6a8287cef27","resolution":{"observed_at":"2026-08-11T04:15:50.284065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.680534Z","title":"Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":"de7e90b3-d5fa-4605-a158-3f7552fc09b8","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.288808Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:d1a32e8bcbccd005c3dc5294c2092e854c8cd0cd183d4bb595cd1def757359fc","observation_id":"4cefecae-94ba-476d-a90d-2929ca3b3499","resolution":{"observed_at":"2026-08-11T04:16:02.685440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.293434Z","title":"Proceedings of the ACM SIGOPS 29th Symposium on Operating Systems Principles , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.293434Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:dd634adf93fd77372f2ac741d9e58271ef6aa9885e54811b074f7a9bfa87760c","observation_id":"40b9e843-fe37-454f-b5c2-c6f94ea407f7","resolution":{"observed_at":"2026-08-11T04:15:50.293434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.298157Z","title":"Proceedings of the 2020 conference on empirical methods in natural language processing: system demonstrations , pages=","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.298157Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:c13689fdb802441fb8cf1fe2f8d7b5f6676263b88b780ec21623a2b7b3b29bf0","observation_id":"ae29f9fd-ef3a-42de-bdfd-e488b92fce03","resolution":{"observed_at":"2026-08-11T04:15:50.298157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.302546Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.302546Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:8c044dc5061ea66acc696f447abaae1bbff3e6d1c456c6e5c57a4f426a76c702","observation_id":"bf9c85f2-7337-4ec9-9e65-c4dde99c28b2","resolution":{"observed_at":"2026-08-11T04:15:50.302546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.632141Z","title":null,"venue":null,"work_id":"d339ca65-9a75-4d2a-8d3b-78b73d6e6781","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.307457Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:1b32cceeefb747201e33ee4fc60fbb795a33a67ca6b362a5825d24996e438254","observation_id":"815fdac4-3bfb-48d9-a829-b0c7a46f8e0d","resolution":{"observed_at":"2026-08-11T04:16:02.636936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.311956Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.311956Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:d3b1265f3a525c7fc307529a87f745ec6fd9b53396240ea98339acf49200383f","observation_id":"b5873800-dee1-4b60-bb4a-915bc9567c28","resolution":{"observed_at":"2026-08-11T04:15:50.311956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.603840Z","title":"2025 , howpublished=","venue":null,"work_id":"0e6458fb-18f5-4ea0-9af2-cf930033d020","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.316888Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:4c0c1f9bdcc7e47e114933a551bf87175985fc39d6770abe9a8c421334caa765","observation_id":"92045abb-e9dd-4d74-9db1-254a9483bd4e","resolution":{"observed_at":"2026-08-11T04:16:02.609581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.587357Z","title":"2025 , howpublished=","venue":null,"work_id":"ae52cbb1-4a10-4191-94af-f022b24750aa","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.321759Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:c60edeb00f7138adece3c750468958f36ff0d9914724e8aaa33f6ed8ecd5b813","observation_id":"25d8b062-34f8-4da2-a0a5-7f10b3c13850","resolution":{"observed_at":"2026-08-11T04:16:02.592933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.570365Z","title":"2025 , howpublished=","venue":null,"work_id":"532d65f5-778f-4546-a196-d67719253180","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.326864Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:af07e15c1af6e9faaf47fdceb28800ddbb1c4c40ddbb611dc1fddc4af6d2bdd6","observation_id":"2aeca693-93f7-4c13-9711-e7c15030698e","resolution":{"observed_at":"2026-08-11T04:16:02.575357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-08-09T11:59:10.408717Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-11T04:15:50.332100Z","title":"arXiv preprint arXiv:2503.01743 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.332100Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:c4b873269d8e08d98a36eb91dbe340787b68359c8c12374c33d60d66cb7a98f5","observation_id":"a1429dfc-cd2d-4df3-95fd-f97f493de6cf","resolution":{"observed_at":"2026-08-11T04:15:50.332100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.551609Z","title":"2024 , howpublished=","venue":null,"work_id":"bcf2a52b-a7fc-4483-92fd-1be6d709e509","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.338259Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:476a936de847b5fe9f01c7b67ddb1d56cea90dc3e682ec0d5b4cf19f539ab0c0","observation_id":"b0dbf0b9-50d1-47eb-bea2-b0314a85e249","resolution":{"observed_at":"2026-08-11T04:16:02.557873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T04:15:50.342633Z","title":"arXiv preprint arXiv:2407.21783 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.342633Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:fc2d53bac01970dedb62973e3140efcd26029d31d28615ff125b19d1db597ba8","observation_id":"cb027bac-a55e-4b95-8d86-4ca64af88f0f","resolution":{"observed_at":"2026-08-11T04:15:50.342633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.348432Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.348432Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:be9dc189321d7a9ff8649f11338160d9ec78a271f2f6b8c2c2b7f70acc7bbf30","observation_id":"bfe618a2-7533-46e0-b512-b01a24e50f98","resolution":{"observed_at":"2026-08-11T04:15:50.348432Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.521467Z","title":"2024 , howpublished=","venue":null,"work_id":"9d66dc51-8fd2-4a2a-b081-8d053a1980a3","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.353587Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:1be5b9348a58a919e4fa8f0af341f0df7923339b2e8b2aba86176ef44936abc5","observation_id":"86918dea-6749-4188-a74f-309f580ed026","resolution":{"observed_at":"2026-08-11T04:16:02.527003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00656","last_updated":"2025-10-08T07:50:45Z","snapshot_observed_at":"2026-08-08T06:58:44.493777Z","submitted_at":"2024-12-31T21:55:10Z","title":"2 OLMo 2 Furious","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00656","snapshot_observed_at":"2026-08-11T04:15:50.358479Z","title":"arXiv preprint arXiv:2501.00656 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.358479Z"},"links":{"cited_paper":"/paper/2501.00656","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:29e59bdbb70fd88b99fa6e1fb1668a3f3b688312e39698df59bdcebd0aaa07ae","observation_id":"2d3a7d1a-332d-47f0-9fcd-836a4ea1f229","resolution":{"observed_at":"2026-08-11T04:15:50.358479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04079","last_updated":"2025-06-16T18:23:31Z","snapshot_observed_at":"2026-08-07T10:45:35.876484Z","submitted_at":"2025-06-04T15:43:31Z","title":"EuroLLM-9B: Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04079","snapshot_observed_at":"2026-08-11T04:15:50.364012Z","title":"arXiv preprint arXiv:2506.04079 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.364012Z"},"links":{"cited_paper":"/paper/2506.04079","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:02566960551c15e2d62c36a79af22d82683676d65c828e7ce167b3f2374f5f28","observation_id":"02edb9c6-4ff4-4f7a-a5b8-de7c709bc20d","resolution":{"observed_at":"2026-08-11T04:15:50.364012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.369439Z","title":"arXiv preprint arXiv:2602.05879 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.369439Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:ad09facb125c9654d4c996b8f9ea71ff847d967d8ee97e12db279bb3e5ec8ac9","observation_id":"dac371bc-fc64-4bfd-a08d-6084ed57bbc0","resolution":{"observed_at":"2026-08-11T04:15:50.369439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.373992Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.373992Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:75be598465b0ef34cca3b86615fcb1d5d0904b6a023a9a80125378430ffc260e","observation_id":"5e22f2d6-7489-4b49-a509-e1d7469e915e","resolution":{"observed_at":"2026-08-11T04:15:50.373992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.490453Z","title":"2025 , url=","venue":null,"work_id":"d8da1811-3fb0-48b4-9435-bae753e1ba8c","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.379740Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:402d5cf27142472c2aff17563e6bf6d80f2ab2c5cf7127322268d5e19c616e8d","observation_id":"8236270b-b58e-477d-877d-5b42023292a8","resolution":{"observed_at":"2026-08-11T04:16:02.496409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00698","last_updated":"2025-04-14T12:37:51Z","snapshot_observed_at":"2026-08-07T16:17:41.511601Z","submitted_at":"2025-04-01T12:08:07Z","title":"Command A: An Enterprise-Ready Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.00698","snapshot_observed_at":"2026-08-11T04:15:50.385072Z","title":"arXiv preprint arXiv:2504.00698 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.385072Z"},"links":{"cited_paper":"/paper/2504.00698","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:63b94368f9f090f7e01aaa79d3b56064799da897d7a2d4971b9a9a17df3af224","observation_id":"2acd2b7d-ab6c-4841-b46f-8803bca34fb3","resolution":{"observed_at":"2026-08-11T04:15:50.385072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.390479Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.390479Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:5429fbfdaf75b12aa845c188179192b41bcfd5a1e97faf109fadb2b7ebb7856a","observation_id":"ecc94888-2081-43de-9c54-c6e32addc108","resolution":{"observed_at":"2026-08-11T04:15:50.390479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10925","last_updated":"2025-08-08T19:24:38Z","snapshot_observed_at":"2026-08-14T02:46:12.121034Z","submitted_at":"2025-08-08T19:24:38Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10925","snapshot_observed_at":"2026-08-11T04:15:50.396377Z","title":"arXiv preprint arXiv:2508.10925 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.396377Z"},"links":{"cited_paper":"/paper/2508.10925","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:4a460fc6b41565296b7a788f213d317344f729067dc070a2b0e5a961af24feba","observation_id":"c4d93b42-8172-4092-b9bb-06e0a4700c80","resolution":{"observed_at":"2026-08-11T04:15:50.396377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.401799Z","title":"2025 , howpublished=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.401799Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:ae55e535365a12569af61b97bcac4d49020abce63ce8fc3051f419828a762979","observation_id":"5ae82857-9e37-4daf-9375-6778ff80e456","resolution":{"observed_at":"2026-08-11T04:15:50.401799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.406896Z","title":"2025 , howpublished=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.406896Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:ffe22894197dc0da491df7f06d44da9a27777f7492208f0217e90709778cd484","observation_id":"92d91374-de2a-4a3d-ae16-6c95610cb7e1","resolution":{"observed_at":"2026-08-11T04:15:50.406896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04261","last_updated":"2024-12-05T15:41:06Z","snapshot_observed_at":"2026-08-13T14:49:22.470549Z","submitted_at":"2024-12-05T15:41:06Z","title":"Aya Expanse: Combining Research Breakthroughs for a New Multilingual Frontier","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04261","snapshot_observed_at":"2026-08-11T04:15:50.413131Z","title":"arXiv preprint arXiv:2412.04261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.413131Z"},"links":{"cited_paper":"/paper/2412.04261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:babfa0b8cb6431d92856cf0937810c50dcb6bb2c5dc3844d0e980474d4bdee9b","observation_id":"98241807-20de-4cd2-a17b-2b23e877d382","resolution":{"observed_at":"2026-08-11T04:15:50.413131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.419053Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.419053Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2d792354a75fc82a5eb69403918f58c2fbebfdee5467b4a2d24eeaf0b272a39d","observation_id":"a5633684-2cb3-4ad7-9de8-c95be94372ae","resolution":{"observed_at":"2026-08-11T04:15:50.419053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.424386Z","title":null,"venue":null,"work_id":"b6b2eadb-32a5-47f0-adbd-2cf6ea3a6588","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.426588Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:5ca52fd2ec018d73007ef715e1a95aba48ccb90de4a93301de2858e6dbd07a45","observation_id":"5c0c811d-1fbb-4654-a570-53fb0d7a5eb7","resolution":{"observed_at":"2026-08-11T04:16:02.430512Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15450","last_updated":"2024-12-19T23:06:01Z","snapshot_observed_at":"2026-08-11T11:23:41.873492Z","submitted_at":"2024-12-19T23:06:01Z","title":"Fietje: An open, efficient LLM for Dutch","version":1},"cited_work":{"arxiv_id":"2412.15450","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.15450","snapshot_observed_at":"2026-08-11T04:15:56.815066Z","title":"Fietje: An open, efficient LLM for Dutch","venue":"cs.CL","work_id":"f0386e41-9f83-4cc2-b622-91d39c8fe8fa","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.431736Z"},"links":{"cited_paper":"/paper/2412.15450","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:0464a3394cb1acb813d2d9e543ddd5ef83875e5eb3f96bf627a392fe4b9103fe","observation_id":"2d4087b1-7719-4aea-baff-0ccca8af4c13","resolution":{"observed_at":"2026-08-11T04:15:56.822815Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.439160Z","title":"arXiv preprint arXiv:2509.14233 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.439160Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:431647742162f36a2b48fc89fb5a8145bf36e0be8e5fc152860ad73cf66fac08","observation_id":"c40d13b9-5766-4a84-a933-6f54a0eb9816","resolution":{"observed_at":"2026-08-11T04:15:50.439160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.447180Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.447180Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:3ab90f9fc4c8a804ab7a692784e0edb715071af2786842e3778cc306af1fa382","observation_id":"4571ed11-2ed4-45dd-b8ca-2f5beb5ba3ff","resolution":{"observed_at":"2026-08-11T04:15:50.447180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.394604Z","title":", author=","venue":null,"work_id":"f794d696-8c81-40c5-977c-5580db35f30b","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.453118Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:308beae7036f6e709ccae1df838e901e0c5c9fbdb7af00d89e429c5eec61dce3","observation_id":"a7d1ec94-7eec-487b-a947-cbd86302aab4","resolution":{"observed_at":"2026-08-11T04:16:02.399928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-14T06:32:32.682623+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-13T23:49:53.171355Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch"},"reference_resolution":{"displayed":77,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":2,"unresolved":46,"verified_exact":0,"verified_fuzzy":28},"total_outbound_references":77},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-14T06:32:32.682623+00:00","source":"crossref"},{"observed_at":"2026-08-14T06:32:18.44784+00:00","source":"retraction_watch"}],"thesis":"As of 14 August 2026, this Paper Citation Record lists 77 of 77 outbound references and 0 inbound Pith citation observations for arXiv:2608.09925."}