{"as_of":"2026-08-20T02:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a4075cb8f998ff464dac74e3c386b880faf723fdbdef57c623cc2c02d54c52b8","coverage":[{"denominator":77,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":77,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T04:15:50.453118Z","state":"measured"},{"denominator":77,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":77,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.09925/citation-record","integrity":"/paper/2608.09925/integrity","json":"/paper/2608.09925/citation-record.json","paper":"/paper/2608.09925"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-11T04:15:50.013344Z","title":"arXiv preprint arXiv:2507.06261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.013344Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:200bb36f7d804fb753eabade8297febcb4d542f1f6eb9a34195019ee80cc8b4d","observation_id":"65bda534-6390-4a18-936c-c885c4249b49","resolution":{"observed_at":"2026-08-11T04:15:50.013344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T04:15:50.020264Z","title":"arXiv preprint arXiv:2303.08774 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.020264Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:dc1cd1d7bd4600308deef2bc7be963b65fa0351f4bd49103128da2eef5c2a7ce","observation_id":"f2ff29d1-b08e-45ba-9da7-4ef0ce1eb598","resolution":{"observed_at":"2026-08-11T04:15:50.020264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-11T04:15:50.026324Z","title":"arXiv preprint arXiv:2302.13971 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.026324Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:84f16f6e1ec6a649db580c99d368bcccff68f2e9d286d0471dc321e3ad022497","observation_id":"a43fa297-80fb-423e-aa19-1fc4cbd1bb1b","resolution":{"observed_at":"2026-08-11T04:15:50.026324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.031881Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.031881Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6c8965480a5c237558e8188312bc1b3e310c22958075e7bd37aa64c0bcb8d886","observation_id":"1feaeb13-ec9b-43fb-8f14-a2d20ffa772c","resolution":{"observed_at":"2026-08-11T04:15:50.031881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.037460Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.037460Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:b1a27772a73ff2ab704bf1a3aa5ac379bcca37386c9d808a36d767157cd6f4c2","observation_id":"27dca7fa-e952-4fbf-a1d5-2a6ee71605ec","resolution":{"observed_at":"2026-08-11T04:15:50.037460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.125637Z","title":"Telecommunications policy , volume=","venue":null,"work_id":"709dabf4-20f5-4193-9019-65b5c4d07fd0","year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.046129Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:b309ba260e0a79aff23b8caace88f3293a5e9019e3f71b0f8645dd1fd78cdb34","observation_id":"59f90afa-1d14-434e-b1be-53896cf2b46e","resolution":{"observed_at":"2026-08-11T04:16:03.131532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.107656Z","title":"Government information quarterly , volume=","venue":null,"work_id":"6be18835-d4e1-40f9-9161-38b9c8a2069d","year":2022},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.057018Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:78d33df17d60ce0e3a94692ce4c73339c20e01366eb9d54063d732686bae2a02","observation_id":"f3d8d368-4e0b-4424-973a-bc438debc1ca","resolution":{"observed_at":"2026-08-11T04:16:03.112621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.091331Z","title":"European Journal of Social Security , volume=","venue":null,"work_id":"6a40adb9-1ed2-45f2-89ad-ce2435187343","year":2021},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.062683Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:419bfc59dd2ca4058de8b3e6851d1ec0581b8be5c968db8c9e986dcc0e704181","observation_id":"509f096f-ebf8-4f9a-8c85-27441266277f","resolution":{"observed_at":"2026-08-11T04:16:03.096441Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.068967Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.068967Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:22627a9897c716caf564cd87af278c31bc69b99c005e2871536376498aa1edd0","observation_id":"c67244c6-792e-4b89-b4e7-957c90311c70","resolution":{"observed_at":"2026-08-11T04:15:50.068967Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.19799","last_updated":"2024-11-29T16:03:14Z","snapshot_observed_at":"2026-08-14T11:28:04.777011Z","submitted_at":"2024-11-29T16:03:14Z","title":"INCLUDE: Evaluating Multilingual Language Understanding with Regional Knowledge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.19799","snapshot_observed_at":"2026-08-11T04:15:50.074768Z","title":"arXiv preprint arXiv:2411.19799 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.074768Z"},"links":{"cited_paper":"/paper/2411.19799","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2b7576f7b19e23365ecd03b2711ad0369f271592eeb8c7cd23ebd6a8e10abaf2","observation_id":"7bef12fe-a67b-48ca-85ce-ffcb8c8d04a5","resolution":{"observed_at":"2026-08-11T04:15:50.074768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.063085Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":"d5973957-136a-40b1-a4e7-8f1db8f38e0b","year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.082059Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:46067b8890c7689ee3ebded6294424e7ca076d75ee49599f0f0a6cf852dae29a","observation_id":"2d3ed7f8-2415-431c-9c42-617aaea06e9a","resolution":{"observed_at":"2026-08-11T04:16:03.068623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.046318Z","title":"Transactions on Machine Learning Research , year=","venue":null,"work_id":"f74c9d33-ac88-4a1e-9d66-85f63743ec85","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.088414Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:5c1c2f00a7048d74ed1507c71f9c5da12d87c1fa8673047b3d20c7523113cfd6","observation_id":"3740ec46-eea5-4231-85a7-a72392c512c1","resolution":{"observed_at":"2026-08-11T04:16:03.051480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.029242Z","title":"GEITje: een groot open Nederlands taalmodel , shorttitle =","venue":null,"work_id":"382aff9f-7070-4604-a5da-d4bfb11ce8fa","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.093926Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:3d37edb75b8667bfb679b640d97e79045008835c7225d71645964fe78e0ba221","observation_id":"547bf310-5a75-41ae-9498-e7edd200614c","resolution":{"observed_at":"2026-08-11T04:16:03.034277Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.099046Z","title":"Proceedings of the 57th annual meeting of the association for computational linguistics , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.099046Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:7fb452491af1c0dfc22e6260e3bf3e89aa0c13ebd5d60feb987635f8ed0cd3de","observation_id":"3ca86f3d-54af-4865-82db-81fe058d544c","resolution":{"observed_at":"2026-08-11T04:15:50.099046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:03.001708Z","title":"Proceedings of the 2024 ACM conference on fairness, accountability, and transparency , pages=","venue":null,"work_id":"8afebc56-528b-43f4-af71-0179cc995eeb","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.105012Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:df9cfb66534f3555ad8071b591d426ef6bd543c9a367d4d22d622b7aff0b9c4f","observation_id":"a818207d-db1c-45ae-823e-da6dc2a7992b","resolution":{"observed_at":"2026-08-11T04:16:03.006747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.09700","last_updated":"2019-11-04T20:37:33Z","snapshot_observed_at":"2026-08-13T09:44:21.186377Z","submitted_at":"2019-10-21T23:57:32Z","title":"Quantifying the Carbon Emissions of Machine Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.09700","snapshot_observed_at":"2026-08-11T04:15:50.111292Z","title":"arXiv preprint arXiv:1910.09700 , year=","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.111292Z"},"links":{"cited_paper":"/paper/1910.09700","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:f02c4f7f283ff38e6520e733a97833f1fd90c2acc7558f23d914b3a06b88a7b4","observation_id":"9d8872d5-a51e-4184-90c1-d75aeee6e346","resolution":{"observed_at":"2026-08-11T04:15:50.111292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.09582","last_updated":"2019-12-19T22:59:26Z","snapshot_observed_at":"2026-08-18T12:45:38.120333Z","submitted_at":"2019-12-19T22:59:26Z","title":"BERTje: A Dutch BERT Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.09582","snapshot_observed_at":"2026-08-11T04:15:50.117536Z","title":"arXiv preprint arXiv:1912.09582 , year=","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.117536Z"},"links":{"cited_paper":"/paper/1912.09582","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:d2bba8b5498fcc16e8350edf9476105783c590bd7c8b37aecdc3db62b864a17e","observation_id":"f9bee712-8313-4257-8894-5256bb14317e","resolution":{"observed_at":"2026-08-11T04:15:50.117536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.984457Z","title":"Findings of the association for computational linguistics: EMNLP 2020 , pages=","venue":null,"work_id":"21f33ebc-4fa8-4c02-b67e-a40f26e101c7","year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.122939Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:f7fe46a6d2c6fea7961ed24969cb2e6096b9e199a83cb3185c2d6970230edf8a","observation_id":"32263682-70fe-4c63-8403-e35e26ade5a1","resolution":{"observed_at":"2026-08-11T04:16:02.990092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13469","last_updated":"2025-01-11T10:20:26Z","snapshot_observed_at":"2026-08-16T13:41:29.336335Z","submitted_at":"2024-06-19T11:50:09Z","title":"Encoder vs Decoder: Comparative Analysis of Encoder and Decoder Language Models on Multilingual NLU Tasks","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13469","snapshot_observed_at":"2026-08-11T04:15:50.128015Z","title":"arXiv preprint arXiv:2406.13469 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.128015Z"},"links":{"cited_paper":"/paper/2406.13469","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:fa6186bae2f5f2b491462fa47187ba8302ff16ae9be4967bd7126a7b818c6aba","observation_id":"dbf5f15b-b203-4a95-8a50-fa5b47e00b3d","resolution":{"observed_at":"2026-08-11T04:15:50.128015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.968153Z","title":null,"venue":null,"work_id":"5881f0de-f050-41c9-919f-50c9136af65f","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.133850Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:cb08eca7bab450b414403b1b3ceb3c440a51cd63bbc6a331b42b7e12a80313cd","observation_id":"82fbc123-7f99-4980-bf17-de6436fc4ce1","resolution":{"observed_at":"2026-08-11T04:16:02.972604Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.952230Z","title":"2025 , month =","venue":null,"work_id":"51981072-14f5-4e03-a20b-d32f588dec8c","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.138883Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:7c8490e86ba490ac72a1971004a41eaba29a074c9cae95747fc6a11852f8d037","observation_id":"db03c3ec-db51-4a11-a72e-c0ad93ffdfb0","resolution":{"observed_at":"2026-08-11T04:16:02.957079Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.936402Z","title":"2025 , month =","venue":null,"work_id":"3526b611-cf4a-4cab-bfaf-4ee7c279177e","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.144427Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:94044af985109385ab4ce71569136487d598719a1f743591dd07f2f895ab689a","observation_id":"20c60e0c-a7fb-47d9-9025-e24bf3f2ac88","resolution":{"observed_at":"2026-08-11T04:16:02.941272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02446","last_updated":"2024-01-27T22:54:52Z","snapshot_observed_at":"2026-08-13T01:25:06.346704Z","submitted_at":"2023-10-03T21:30:56Z","title":"Low-Resource Languages Jailbreak GPT-4","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02446","snapshot_observed_at":"2026-08-11T04:15:50.149734Z","title":"arXiv preprint arXiv:2310.02446 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.149734Z"},"links":{"cited_paper":"/paper/2310.02446","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:19ec293283a40c5cb17e997d20467212d9384be5357008733cc938ea672e3f85","observation_id":"749fd45f-f6c6-445f-80e0-106a461a5a3b","resolution":{"observed_at":"2026-08-11T04:15:50.149734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.919765Z","title":"Pending submission","venue":null,"work_id":"ef324bb1-5d6d-4215-804c-39a60fc15085","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.155277Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:ad0d22e49a24d94bc5e33b59003dbf225a93df216471479a6f30283a4abbf4f1","observation_id":"2b544e77-1e70-46c8-901d-9253df04b52e","resolution":{"observed_at":"2026-08-11T04:16:02.924779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.903497Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"2a11134e-1385-448a-89b3-f8de2fb262d0","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.161741Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2d5238320afe2b9713e764a03ae25498b0d6df791bd75f6252956a72789888c9","observation_id":"287dbf20-d41e-4dc2-b085-5002b5ff2516","resolution":{"observed_at":"2026-08-11T04:16:02.908434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13261","last_updated":"2024-07-08T18:29:58Z","snapshot_observed_at":"2026-08-16T13:41:34.849705Z","submitted_at":"2024-06-19T06:46:59Z","title":"BeHonest: Benchmarking Honesty in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13261","snapshot_observed_at":"2026-08-11T04:15:50.166929Z","title":"arXiv preprint arXiv:2406.13261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.166929Z"},"links":{"cited_paper":"/paper/2406.13261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:3640637b7192789a21fc1623f8cb55c617e005837404e57b873c727670852fb7","observation_id":"fc86b4ff-f50e-443f-b78b-adfa071959e4","resolution":{"observed_at":"2026-08-11T04:15:50.166929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.01768","last_updated":"2023-01-05T07:13:13Z","snapshot_observed_at":"2026-08-16T16:04:18.853456Z","submitted_at":"2023-01-05T07:13:13Z","title":"The political ideology of conversational AI: Converging evidence on ChatGPT's pro-environmental, left-libertarian orientation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.01768","snapshot_observed_at":"2026-08-11T04:15:50.172952Z","title":"arXiv preprint arXiv:2301.01768 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.172952Z"},"links":{"cited_paper":"/paper/2301.01768","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:e4c5ee9ab1a4f83a30d1751f7787c1107a5b6a6684034c1ab39dde3a71d51905","observation_id":"2d9a8600-25bd-4edc-8688-0bafe4b0586e","resolution":{"observed_at":"2026-08-11T04:15:50.172952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.178616Z","title":"PloS one , volume=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.178616Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2f1d238344c0fcffc701a379c26414800a81a1052cdfb69c73873d41a8c2ebab","observation_id":"5f15cbef-9b6d-4f9d-ba47-2041c409fc0c","resolution":{"observed_at":"2026-08-11T04:15:50.178616Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.876576Z","title":"Proceedings of the AAAI/ACM Conference on AI, Ethics, and Society , year=","venue":null,"work_id":"3d87cb07-7846-4597-a0aa-0d2678e9eef1","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.184072Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:cbe38ad9901570719507249d6eece1f0ccff449f089814c55149e3dbf49bebc8","observation_id":"224fa6fa-39c4-4536-9b8b-a3b8aff85088","resolution":{"observed_at":"2026-08-11T04:16:02.882505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.190324Z","title":"arXiv preprint arXiv:2601.08785 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.190324Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:d95181f4765a3e622b101f0f6528d3d8591399736cee63cd7de6dca928a835ee","observation_id":"22933243-28db-4a31-b839-6d2ef8753e7a","resolution":{"observed_at":"2026-08-11T04:15:50.190324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-08-14T19:36:07.505691Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-11T04:15:50.195330Z","title":"arXiv preprint arXiv:1803.05457 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.195330Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2f19a9aa2bce8c11a9a6a74ee59b7f04af9b0671681cf20db7603d9d1031576d","observation_id":"ea88fa7e-2a71-4bc5-8928-94ea95e64a67","resolution":{"observed_at":"2026-08-11T04:15:50.195330Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.14992","last_updated":"2024-05-26T22:27:23Z","snapshot_observed_at":"2026-08-17T02:59:38.954198Z","submitted_at":"2024-02-22T22:05:23Z","title":"tinyBenchmarks: evaluating LLMs with fewer examples","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.14992","snapshot_observed_at":"2026-08-11T04:15:50.201032Z","title":"arXiv preprint arXiv:2402.14992 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.201032Z"},"links":{"cited_paper":"/paper/2402.14992","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6d990e161ecb56b2dbd4922d0cf1c0f15f0efee414bcc325d055350eadd526c2","observation_id":"f6658ab6-13d1-4a6b-9915-7a3c8ce8e323","resolution":{"observed_at":"2026-08-11T04:15:50.201032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.859160Z","title":"Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing: System Demonstrations , pages=","venue":null,"work_id":"8a1675ac-9c66-4d7b-9d28-703fb9c72c3f","year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.207001Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:fd59c9232b1bc5b4a25618f1c04a910cf936563156bb273c1e8de4dc57bcfcc0","observation_id":"2f9ac80b-b3ab-4b01-ad5f-2bfc61acced6","resolution":{"observed_at":"2026-08-11T04:16:02.864600Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.213983Z","title":"Proceedings of the 2020 conference on empirical methods in natural language processing (EMNLP) , pages=","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.213983Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:1cc8c234946d8234ca584f614c331625d240490f049dd143fde351890b454a52","observation_id":"6aa152bb-944e-4066-89a8-6c6ba9f70849","resolution":{"observed_at":"2026-08-11T04:15:50.213983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.831811Z","title":"Proceedings of the 15th International Conference on Recent Advances in Natural Language Processing-Natural Language Processing in the Generative AI Era , pages=","venue":null,"work_id":"062f6bde-27d3-449a-a8a6-199fc4f459c2","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.219523Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:5a94ba9028afca2b77c97e9a4bbae448fdd2e5c823b01272bfbbce586a80ceeb","observation_id":"159dd60e-07d8-4329-8bb7-fd9cddd64f00","resolution":{"observed_at":"2026-08-11T04:16:02.837311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.815033Z","title":null,"venue":null,"work_id":"1ac9ab6e-1f5e-4633-8bc9-6e152c91ebaa","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.224827Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:9906d3a73585b198b4b1e56476a4c801a7abca7e3d6fe5a8c9c3c1263c2f46a6","observation_id":"9fe2a5e1-15c8-42dd-bd8a-2189dac4fd9c","resolution":{"observed_at":"2026-08-11T04:16:02.820106Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.230578Z","title":"Proceedings of the 60th annual meeting of the association for computational linguistics (volume 1: long papers) , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.230578Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:de3970eb020dadb2d53d058e113dfeca45d6696de1fec3f8fbaa9f2aae8eabbf","observation_id":"6a7f0154-340c-4fdc-b1a5-0465ddbcd5a4","resolution":{"observed_at":"2026-08-11T04:15:50.230578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.235724Z","title":"Findings of the Association for Computational Linguistics: ACL 2022 , pages=","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.235724Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:732f824e69ed566e0159a72e5aed8cf21a883bc4c75ee62ac8ce5cb77ce24577","observation_id":"ae4fd28e-bb08-49ed-a7b3-72da331302df","resolution":{"observed_at":"2026-08-11T04:15:50.235724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-11T04:15:50.241431Z","title":"arXiv preprint arXiv:2009.03300 , year=","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.241431Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:953c7d78f3e93b2db1b36c7c9380c389146c0c75453f8150f0ae5657a6a193a4","observation_id":"c2601e12-de8c-4528-9db1-449b9b46df53","resolution":{"observed_at":"2026-08-11T04:15:50.241431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07243","last_updated":"2024-07-17T08:49:22Z","snapshot_observed_at":"2026-08-16T13:44:11.096599Z","submitted_at":"2024-06-11T13:23:14Z","title":"MBBQ: A Dataset for Cross-Lingual Comparison of Stereotypes in Generative LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07243","snapshot_observed_at":"2026-08-11T04:15:50.247552Z","title":"arXiv preprint arXiv:2406.07243 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.247552Z"},"links":{"cited_paper":"/paper/2406.07243","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:501ff3781019bd43863490b50b00f0780b8100fd02ced9b3351f03cb3a1a8abc","observation_id":"04e13413-6b6c-4120-a6b7-2cbf99295609","resolution":{"observed_at":"2026-08-11T04:15:50.247552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.775834Z","title":"Proceedings of the 2023 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":"5e02034a-09af-47bd-828b-04e6ada351ac","year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.254727Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:98804cc86acaaadd7e5dab4f9120c451fadc24c006642e0b0982a6405cece7bd","observation_id":"1f9e392f-045f-49c1-8ec5-a8beab3e9a77","resolution":{"observed_at":"2026-08-11T04:16:02.781074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.758876Z","title":"Proceedings of the 2024 joint international conference on computational linguistics, language resources and evaluation (lrec-coling 2024) , pages=","venue":null,"work_id":"6c4d226a-f84d-4a81-a826-2d429e68b56a","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.261054Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:ddf1a9841345bc00c6c60757fa47cd131cd1d9c07ac09379495773b341ddea5e","observation_id":"fe66995b-c666-4f1c-bd04-f87974963e67","resolution":{"observed_at":"2026-08-11T04:16:02.764354Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.742505Z","title":"Online:< https://ticclops","venue":null,"work_id":"94ea9b85-c964-4f69-97ff-3ac49258747a","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.267364Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:675fff28a063bcb275891532ca24b6ca9f2082bcf47e3a0b5fe0a04006b14230","observation_id":"43498d15-060c-4f3a-b39e-2cf7f9b0c2b9","resolution":{"observed_at":"2026-08-11T04:16:02.747690Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.726276Z","title":"CLARIN Annual Conference Proceedings , pages=","venue":null,"work_id":"d235756f-c0c8-4b67-8ebc-b4c9f2430156","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.273694Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:b25ba96cffc7d79ed310cf9661fc456a1b876e149befa3518d3fddccfe4a5102","observation_id":"98e3bea8-9de1-4ead-870d-bab5db8c0dcf","resolution":{"observed_at":"2026-08-11T04:16:02.731379Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.706956Z","title":"Proceedings of the 2018 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":"93e751a7-e94d-461b-a4a9-773b15e88cea","year":2018},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.279029Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:d1f9d7f2dd2244eedb2463352bf60aee77915707655fc069fa0f98f66f9f0952","observation_id":"ef2f24ce-dd09-4404-be85-34d247c262d6","resolution":{"observed_at":"2026-08-11T04:16:02.712268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.284065Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.284065Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:06408f15dcf55f381c2fbebecbc1c1a755b80c054b9f2b12a4382633bb8aa134","observation_id":"702c8270-c0fb-4578-b5ee-e6a8287cef27","resolution":{"observed_at":"2026-08-11T04:15:50.284065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.680534Z","title":"Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":"de7e90b3-d5fa-4605-a158-3f7552fc09b8","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.288808Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:68f43eb0341c9c53b762c7b5b9677e99a2381f6a35acbe22a617dc26de9f5eb9","observation_id":"4cefecae-94ba-476d-a90d-2929ca3b3499","resolution":{"observed_at":"2026-08-11T04:16:02.685440Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.293434Z","title":"Proceedings of the ACM SIGOPS 29th Symposium on Operating Systems Principles , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.293434Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:2e0b291a36081f70bb572ee0ae6cd6c6b51a34ef97eead1a0c4cf5c41e41943e","observation_id":"40b9e843-fe37-454f-b5c2-c6f94ea407f7","resolution":{"observed_at":"2026-08-11T04:15:50.293434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.298157Z","title":"Proceedings of the 2020 conference on empirical methods in natural language processing: system demonstrations , pages=","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.298157Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6db548e29458ec537ea2987ad5547b45406abd2ff741d47af17baeee69791b12","observation_id":"ae29f9fd-ef3a-42de-bdfd-e488b92fce03","resolution":{"observed_at":"2026-08-11T04:15:50.298157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.302546Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.302546Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:3b7c9eeb25d596f1ca757000a165edaf91161e4473814cdf124dd972b709b096","observation_id":"bf9c85f2-7337-4ec9-9e65-c4dde99c28b2","resolution":{"observed_at":"2026-08-11T04:15:50.302546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.632141Z","title":null,"venue":null,"work_id":"d339ca65-9a75-4d2a-8d3b-78b73d6e6781","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.307457Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:a783762d4d45e29f6d027aa38da4bd13781217d097e977c19c1c1e4177231e23","observation_id":"815fdac4-3bfb-48d9-a829-b0c7a46f8e0d","resolution":{"observed_at":"2026-08-11T04:16:02.636936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.311956Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.311956Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:36d72c3c2432b9e61db7b84d5f241699eb411541aeb20be0af1f9d7bdb5831cc","observation_id":"b5873800-dee1-4b60-bb4a-915bc9567c28","resolution":{"observed_at":"2026-08-11T04:15:50.311956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.603840Z","title":"2025 , howpublished=","venue":null,"work_id":"0e6458fb-18f5-4ea0-9af2-cf930033d020","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.316888Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:46ac5c1cb7fd2f937f82dc257f4f272b9c92c8727a19286d7bedb53e2b2d2b09","observation_id":"92045abb-e9dd-4d74-9db1-254a9483bd4e","resolution":{"observed_at":"2026-08-11T04:16:02.609581Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.587357Z","title":"2025 , howpublished=","venue":null,"work_id":"ae52cbb1-4a10-4191-94af-f022b24750aa","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.321759Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:35bf65daffc29cdc0fd2afa19fa15263f8f6e057035218f60c54600ff65dc1e5","observation_id":"25d8b062-34f8-4da2-a0a5-7f10b3c13850","resolution":{"observed_at":"2026-08-11T04:16:02.592933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.570365Z","title":"2025 , howpublished=","venue":null,"work_id":"532d65f5-778f-4546-a196-d67719253180","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.326864Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:285afa2fd10b68bd0e2bf993205786eff7a03a477de41c00fedd6635e2b5a1f9","observation_id":"2aeca693-93f7-4c13-9711-e7c15030698e","resolution":{"observed_at":"2026-08-11T04:16:02.575357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01743","last_updated":"2025-03-07T09:05:58Z","snapshot_observed_at":"2026-08-15T22:57:45.773661Z","submitted_at":"2025-03-03T17:05:52Z","title":"Phi-4-Mini Technical Report: Compact yet Powerful Multimodal Language Models via Mixture-of-LoRAs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01743","snapshot_observed_at":"2026-08-11T04:15:50.332100Z","title":"arXiv preprint arXiv:2503.01743 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.332100Z"},"links":{"cited_paper":"/paper/2503.01743","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:e1157fbd5399d43aadef5731f0e9d2822366d9f94a9f0ec7c675ea8383803fd7","observation_id":"a1429dfc-cd2d-4df3-95fd-f97f493de6cf","resolution":{"observed_at":"2026-08-11T04:15:50.332100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.551609Z","title":"2024 , howpublished=","venue":null,"work_id":"bcf2a52b-a7fc-4483-92fd-1be6d709e509","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.338259Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:21b68b9f2fe1bb23517eb948d348eea070ad982426e9e64a71201c7a498b4b93","observation_id":"b0dbf0b9-50d1-47eb-bea2-b0314a85e249","resolution":{"observed_at":"2026-08-11T04:16:02.557873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T04:15:50.342633Z","title":"arXiv preprint arXiv:2407.21783 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.342633Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:a2730382873912b059e05bb7a68e44836798a2dc9ba5abf12a74edb5607f8d47","observation_id":"cb027bac-a55e-4b95-8d86-4ca64af88f0f","resolution":{"observed_at":"2026-08-11T04:15:50.342633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.348432Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.348432Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:db7460be20514997556a99e8177c0331d336ad21ff1ea36b046ef6c6a434d06b","observation_id":"bfe618a2-7533-46e0-b512-b01a24e50f98","resolution":{"observed_at":"2026-08-11T04:15:50.348432Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.521467Z","title":"2024 , howpublished=","venue":null,"work_id":"9d66dc51-8fd2-4a2a-b081-8d053a1980a3","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.353587Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:29627b5ba40621c7c74adc40fe0535308d4fa6eec3dd8d56cf2c2f68fcf140f1","observation_id":"86918dea-6749-4188-a74f-309f580ed026","resolution":{"observed_at":"2026-08-11T04:16:02.527003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00656","last_updated":"2025-10-08T07:50:45Z","snapshot_observed_at":"2026-08-17T13:58:40.683829Z","submitted_at":"2024-12-31T21:55:10Z","title":"2 OLMo 2 Furious","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.00656","snapshot_observed_at":"2026-08-11T04:15:50.358479Z","title":"arXiv preprint arXiv:2501.00656 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.358479Z"},"links":{"cited_paper":"/paper/2501.00656","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:1a4f664f7c76f97dcbc712638aca231999212645c5003d85270c73452a929ccf","observation_id":"2d3a7d1a-332d-47f0-9fcd-836a4ea1f229","resolution":{"observed_at":"2026-08-11T04:15:50.358479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04079","last_updated":"2025-06-16T18:23:31Z","snapshot_observed_at":"2026-08-18T03:28:53.509446Z","submitted_at":"2025-06-04T15:43:31Z","title":"EuroLLM-9B: Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.04079","snapshot_observed_at":"2026-08-11T04:15:50.364012Z","title":"arXiv preprint arXiv:2506.04079 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.364012Z"},"links":{"cited_paper":"/paper/2506.04079","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:893c023832a0578e3652917d178ecdb59fbb6b7e59c9e965b7a629aca387a5e9","observation_id":"02edb9c6-4ff4-4f7a-a5b8-de7c709bc20d","resolution":{"observed_at":"2026-08-11T04:15:50.364012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.369439Z","title":"arXiv preprint arXiv:2602.05879 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.369439Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:65ddd0044addbddaf8306ec460ef1c63305c443c41e51ca4f92e9fa77ea5b59f","observation_id":"dac371bc-fc64-4bfd-a08d-6084ed57bbc0","resolution":{"observed_at":"2026-08-11T04:15:50.369439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.373992Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.373992Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:e4d3cf8f6acb569bbb3f03ace5bc088cd75edffaf9d7e4be2323d901b2a71edb","observation_id":"5e22f2d6-7489-4b49-a509-e1d7469e915e","resolution":{"observed_at":"2026-08-11T04:15:50.373992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.490453Z","title":"2025 , url=","venue":null,"work_id":"d8da1811-3fb0-48b4-9435-bae753e1ba8c","year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.379740Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:8a0df1dd643d8273dd178132d2fe5a580b5a93b075f901da565e461fe8e92647","observation_id":"8236270b-b58e-477d-877d-5b42023292a8","resolution":{"observed_at":"2026-08-11T04:16:02.496409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00698","last_updated":"2025-04-14T12:37:51Z","snapshot_observed_at":"2026-08-17T05:47:31.276123Z","submitted_at":"2025-04-01T12:08:07Z","title":"Command A: An Enterprise-Ready Large Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.00698","snapshot_observed_at":"2026-08-11T04:15:50.385072Z","title":"arXiv preprint arXiv:2504.00698 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.385072Z"},"links":{"cited_paper":"/paper/2504.00698","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:6ec0cbf61c3aaad5806b61d9c0965c0498ce576badc991d6e9c927a73d1092e2","observation_id":"2acd2b7d-ab6c-4841-b46f-8803bca34fb3","resolution":{"observed_at":"2026-08-11T04:15:50.385072Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.390479Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.390479Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:758cf295c1cc61ded55097b32f50e709a106efa4834f43fbd03bffdf4b21a04f","observation_id":"ecc94888-2081-43de-9c54-c6e32addc108","resolution":{"observed_at":"2026-08-11T04:15:50.390479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.10925","last_updated":"2025-08-08T19:24:38Z","snapshot_observed_at":"2026-08-14T02:46:12.121034Z","submitted_at":"2025-08-08T19:24:38Z","title":"gpt-oss-120b & gpt-oss-20b Model Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.10925","snapshot_observed_at":"2026-08-11T04:15:50.396377Z","title":"arXiv preprint arXiv:2508.10925 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.396377Z"},"links":{"cited_paper":"/paper/2508.10925","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:d8abc8b9b15ba0d5dc93cc4249095eadf5b7725c543414f9831d3ecb93d8f14d","observation_id":"c4d93b42-8172-4092-b9bb-06e0a4700c80","resolution":{"observed_at":"2026-08-11T04:15:50.396377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.401799Z","title":"2025 , howpublished=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.401799Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:359232ffb106043e8236285b841d091fd9271b6169e21a8c87140e2ad75d33f6","observation_id":"5ae82857-9e37-4daf-9375-6778ff80e456","resolution":{"observed_at":"2026-08-11T04:15:50.401799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.406896Z","title":"2025 , howpublished=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.406896Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:89f64a16587460acf54339ee72a4177ad5811bec5bb4e6e5fe46f5c4695b725e","observation_id":"92d91374-de2a-4a3d-ae16-6c95610cb7e1","resolution":{"observed_at":"2026-08-11T04:15:50.406896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04261","last_updated":"2024-12-05T15:41:06Z","snapshot_observed_at":"2026-08-13T14:49:22.470549Z","submitted_at":"2024-12-05T15:41:06Z","title":"Aya Expanse: Combining Research Breakthroughs for a New Multilingual Frontier","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04261","snapshot_observed_at":"2026-08-11T04:15:50.413131Z","title":"arXiv preprint arXiv:2412.04261 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.413131Z"},"links":{"cited_paper":"/paper/2412.04261","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:0386938c8e83689c77c7fe159c33ce09c2749716ea6e5ab0bdb3b6972ed6d45f","observation_id":"98241807-20de-4cd2-a17b-2b23e877d382","resolution":{"observed_at":"2026-08-11T04:15:50.413131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.419053Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.419053Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:9f4f390e654d72eeea5a0dcdf9cd3c3b48ea5ce920438f24af22117cb2d82872","observation_id":"a5633684-2cb3-4ad7-9de8-c95be94372ae","resolution":{"observed_at":"2026-08-11T04:15:50.419053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.424386Z","title":null,"venue":null,"work_id":"b6b2eadb-32a5-47f0-adbd-2cf6ea3a6588","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.426588Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:28f0b943f6fa62ab20ac5b3780ffbc2514c6bb5b53b1cf08bea10aeb8ef47581","observation_id":"5c0c811d-1fbb-4654-a570-53fb0d7a5eb7","resolution":{"observed_at":"2026-08-11T04:16:02.430512Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15450","last_updated":"2024-12-19T23:06:01Z","snapshot_observed_at":"2026-08-16T18:59:05.613139Z","submitted_at":"2024-12-19T23:06:01Z","title":"Fietje: An open, efficient LLM for Dutch","version":1},"cited_work":{"arxiv_id":"2412.15450","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.15450","snapshot_observed_at":"2026-08-11T04:15:56.815066Z","title":"Fietje: An open, efficient LLM for Dutch","venue":"cs.CL","work_id":"f0386e41-9f83-4cc2-b622-91d39c8fe8fa","year":2024},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.431736Z"},"links":{"cited_paper":"/paper/2412.15450","citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:8f4f5481a76351b8e9f24c564bd14e55632e447c92066e7f7f7d3b7f9746e4d8","observation_id":"2d4087b1-7719-4aea-baff-0ccca8af4c13","resolution":{"observed_at":"2026-08-11T04:15:56.822815Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.439160Z","title":"arXiv preprint arXiv:2509.14233 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.439160Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:fec5665eea1a9e285288c877965709e034b436e8e852a7e07f9159084147c62b","observation_id":"c40d13b9-5766-4a84-a933-6f54a0eb9816","resolution":{"observed_at":"2026-08-11T04:15:50.439160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:15:50.447180Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.447180Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:268a86dcf9054fc22a7f78acd46a7443f0cbf34531ff22b72790066eb786d201","observation_id":"4571ed11-2ed4-45dd-b8ca-2f5beb5ba3ff","resolution":{"observed_at":"2026-08-11T04:15:50.447180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T04:16:02.394604Z","title":", author=","venue":null,"work_id":"f794d696-8c81-40c5-977c-5580db35f30b","year":null},"citing_paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-11T04:15:50.453118Z"},"links":{"citing_paper":"/paper/2608.09925"},"observation_digest":"sha256:e79ca5c99d5b439c98a66eb2a94165e3e30c2ab63a2b814940f75c73a438fd22","observation_id":"a7d1ec94-7eec-487b-a947-cbd86302aab4","resolution":{"observed_at":"2026-08-11T04:16:02.399928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.09925","last_updated":"2026-08-10T17:59:05Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-18T09:30:26.595553Z","submitted_at":"2026-08-10T17:59:05Z","title":"From Values to Benchmarks: Evaluating Large Language Models for Governmental Use in Dutch"},"reference_resolution":{"displayed":77,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":2,"unresolved":46,"verified_exact":0,"verified_fuzzy":28},"total_outbound_references":77},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 77 of 77 outbound references and 0 inbound Pith citation observations for arXiv:2608.09925."}