{"as_of":"2026-08-23T05:18:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5122f99f13c965029b3e119f49a03cb4a6791667c60790c6bec59992930dea2f","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T19:12:55.209578Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:42:53.504979Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-22T03:41:00.103963Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-07T05:42:53.504979Z","title":"arXiv:2411.10939 [cs.CY] https://arxiv.org/abs/2411.10939","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.07282","last_updated":"2025-06-08T21:02:33Z","snapshot_observed_at":"2026-08-19T18:27:08.015980Z","submitted_at":"2025-06-08T21:02:33Z","title":"Adultification Bias in LLMs and Text-to-Image Models","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T05:42:53.504979Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2506.07282"},"observation_digest":"sha256:1d8b3d6bf0c19ec787aece9a4cc2149c76ecca294f49ac0727833454f74267d0","observation_id":"d9a316a6-17c7-4f4c-8413-eac0b080776c","resolution":{"observed_at":"2026-08-07T05:42:53.504979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-07T05:27:56.330863Z","title":"F., Wang, A., Barocas, S., Chouldechova, A., Atalla, C., Blodgett, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07962","last_updated":"2025-06-09T17:37:18Z","snapshot_observed_at":"2026-08-19T14:09:29.261297Z","submitted_at":"2025-06-09T17:37:18Z","title":"Correlated Errors in Large Language Models","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T05:27:56.330863Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2506.07962"},"observation_digest":"sha256:5f60c18e16427b22995df024c884f5208f5afe138a58a209ffee69fe1d7627cd","observation_id":"dc84554f-f5ac-46c7-9ab3-8b4ce16d6715","resolution":{"observed_at":"2026-08-07T05:27:56.330863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-06T20:24:11.241009Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02819","last_updated":"2025-08-19T13:24:31Z","snapshot_observed_at":"2026-08-06T21:37:13.724015Z","submitted_at":"2025-07-03T17:33:24Z","title":"Measurement as Bricolage: Examining How Data Scientists Construct Target Variables for Predictive Modeling Tasks","version":3},"reference_index":140,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:11.241009Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2507.02819"},"observation_digest":"sha256:cbfa3500ac6f2925ffd229a26bc6aa85ac1259804296e0c26a20edee9d801cc0","observation_id":"5d511600-3fcc-4607-90f9-c6c2637fb8cc","resolution":{"observed_at":"2026-08-06T20:24:11.241009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-05T16:40:07.216514Z","title":"arXiv:2411.10939 [cs]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.18076","last_updated":"2025-08-27T21:24:13Z","snapshot_observed_at":"2026-08-15T22:20:39.258771Z","submitted_at":"2025-08-25T14:43:10Z","title":"Neither Valid nor Reliable? Investigating the Use of LLMs as Judges","version":2},"reference_index":112,"source":"pdf_text","source_observed_at":"2026-08-05T16:40:07.216514Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2508.18076"},"observation_digest":"sha256:864a7bb83c8735b8f0f9a23718f4be5c563040c981152434abf7c7c4154c3e70","observation_id":"71c16f38-00e1-4ca8-9015-6337f292eea6","resolution":{"observed_at":"2026-08-05T16:40:07.216514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-05T14:38:55.710426Z","title":"Feder Cooper, Angelina Wang, Solon Barocas, Alexandra Chouldechova, Chad Atalla, Su Lin Blodgett, Emily Corvi, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21036","last_updated":"2025-08-28T17:40:42Z","snapshot_observed_at":"2026-08-16T03:45:58.625345Z","submitted_at":"2025-08-28T17:40:42Z","title":"Understanding, Protecting, and Augmenting Human Cognition with Generative AI: A Synthesis of the CHI 2025 Tools for Thought Workshop","version":1},"reference_index":127,"source":"pdf_text","source_observed_at":"2026-08-05T14:38:55.710426Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2508.21036"},"observation_digest":"sha256:4826333d14c4fa40e4fffecf0478a9df332fbc257f3af895268b586c6d24feb7","observation_id":"6f3764f2-2522-4499-89c9-7d0de8f765e4","resolution":{"observed_at":"2026-08-05T14:38:55.710426Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-04T20:35:57.153507Z","title":"arXiv:2411.10939 [cs]","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08494","last_updated":"2025-09-10T11:10:10Z","snapshot_observed_at":"2026-08-16T15:05:34.983903Z","submitted_at":"2025-09-10T11:10:10Z","title":"HumanAgencyBench: Scalable Evaluation of Human Agency Support in AI Assistants","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-04T20:35:57.153507Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2509.08494"},"observation_digest":"sha256:300ad94d6b9e1bbc9f0ab4a4ac384e10a0bc9e62c4a21c00c89c9b72036351b2","observation_id":"ef6e2618-610c-446f-948e-752f28d4d8e0","resolution":{"observed_at":"2026-08-04T20:35:57.153507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":"2411.10939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluating generative ai systems is a social science measurement challenge.arXiv preprint arXiv:2411.10939","venue":null,"work_id":"9c0a78a9-0fef-43bf-8df6-66b6b69399a5","year":2024},"citing_paper":{"arxiv_id":"2604.25580","last_updated":"2026-04-28T12:49:54Z","snapshot_observed_at":"2026-08-16T17:51:20.812560Z","submitted_at":"2026-04-28T12:49:54Z","title":"Bye Bye Perspective API: Lessons for Measurement Infrastructure in NLP, CSS and LLM Evaluation","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-07T16:09:27.944432Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2604.25580"},"observation_digest":"sha256:b020b7c7ded2988dbd3af79ed538f1ef20fde0da6eba8623712a2315fdda69d4","observation_id":"ea55a5d3-78b6-4562-babf-f5bdb69bda7b","resolution":{"observed_at":"2026-05-11T23:51:21.097511Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":"2411.10939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluating generative ai systems is a social science measurement challenge.arXiv preprint arXiv:2411.10939","venue":null,"work_id":"9c0a78a9-0fef-43bf-8df6-66b6b69399a5","year":2024},"citing_paper":{"arxiv_id":"2605.10834","last_updated":"2026-07-28T07:30:21Z","snapshot_observed_at":"2026-08-14T09:22:43.918837Z","submitted_at":"2026-05-11T16:50:00Z","title":"From Controlled to the Wild: Evaluation of Pentesting Agents for the Real-World","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-12T03:34:55.538935Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2605.10834"},"observation_digest":"sha256:2c110bc935391931b6abfe016490a1dad79d7fcbbd423d70c81135cdc88f20df","observation_id":"2456370c-6165-4ec4-9876-0f4431a0d06b","resolution":{"observed_at":"2026-05-12T07:16:27.754116Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-08-02T14:22:26.223129Z","title":"Cooper, Angelina Wang, Solon Barocas, Alexandra Chouldechova, Chad Atalla, Su Lin Blodgett, Emily Corvi, P","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2605.10834","last_updated":"2026-07-28T07:30:21Z","snapshot_observed_at":"2026-08-14T09:22:43.918837Z","submitted_at":"2026-05-11T16:50:00Z","title":"From Controlled to the Wild: Evaluation of Pentesting Agents for the Real-World","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T14:22:26.223129Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2605.10834"},"observation_digest":"sha256:c226e02f4b049cb01999c4f48fd61bcef299deeab1827de764947f9512790f69","observation_id":"e537999c-a49b-449d-b305-8ca37468295c","resolution":{"observed_at":"2026-08-02T14:22:26.223129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":"2411.10939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluating generative ai systems is a social science measurement challenge.arXiv preprint arXiv:2411.10939","venue":null,"work_id":"9c0a78a9-0fef-43bf-8df6-66b6b69399a5","year":2024},"citing_paper":{"arxiv_id":"2605.15990","last_updated":"2026-05-15T14:21:52Z","snapshot_observed_at":"2026-08-16T12:49:13.103698Z","submitted_at":"2026-05-15T14:21:52Z","title":"Defining Cultural Capabilities for AI Evaluation: A Taxonomy Grounded in Intercultural Communication Theory","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-20T19:40:59.534448Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2605.15990"},"observation_digest":"sha256:b2bfafdaaa96a75fd22bbc554d45d35a469d9addefed57d7571b2d60c7db1986","observation_id":"48a4c397-c291-4af2-ab40-b86e1b1d9eab","resolution":{"observed_at":"2026-05-20T19:43:43.866041Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"cited_work":{"arxiv_id":"2411.10939","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.10939","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluating generative ai systems is a social science measurement challenge.arXiv preprint arXiv:2411.10939","venue":null,"work_id":"9c0a78a9-0fef-43bf-8df6-66b6b69399a5","year":2024},"citing_paper":{"arxiv_id":"2605.22612","last_updated":"2026-05-21T15:27:58Z","snapshot_observed_at":"2026-08-17T12:41:09.535088Z","submitted_at":"2026-05-21T15:27:58Z","title":"Healthcare LLM Benchmarks Are Only as Good as Their Explicit Assumptions","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-22T03:40:25.093487Z"},"links":{"cited_paper":"/paper/2411.10939","citing_paper":"/paper/2605.22612"},"observation_digest":"sha256:6f4e6e84bc72be91f05fd2a6bbb3b33e0a67dc1ff382b403935495196f3cc92b","observation_id":"36e9d93d-e554-4dd3-852f-f020103e3331","resolution":{"observed_at":"2026-05-22T03:41:00.107255Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.10939/citation-record","integrity":"/paper/2411.10939/integrity","json":"/paper/2411.10939/citation-record.json","paper":"/paper/2411.10939"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"answer/2801939","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.560355Z","title":"YouTube Hate Speech Policy","venue":null,"work_id":"363f256f-b7dd-422e-825d-9d07e02382d3","year":null},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.096303Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:4b8781655b5df4f68d03a8ebe1b384dd603f8ae9a7ed5bc9430470a3d8f60c8b","observation_id":"9c515841-4d64-40e6-9d02-41c64f387539","resolution":{"observed_at":"2026-08-12T19:12:55.571058Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.101173Z","title":"Measurement validity: A shared standard for qualitative and quantitative research","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.101173Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:da63d67571cff15a843d4cbf1de3060d1fd8e55dac623a83b277a58d04853f19","observation_id":"91d834e9-b16e-4393-9a37-040723445a03","resolution":{"observed_at":"2026-08-12T19:12:55.101173Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.773660Z","title":"Content analysis in communication research, 1952","venue":null,"work_id":"2c05f7e7-0f6a-493b-ba0a-a8b8755c3d75","year":1952},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.105802Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:5d61e3655fff2fb1c8c21208d9976362a06d47e07acca6a3ffe81f7e4851f98e","observation_id":"b064fff7-1647-44f2-88fe-f573527e55fc","resolution":{"observed_at":"2026-08-12T19:12:55.779337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.110397Z","title":"Making Intelligence: Ethical Values in IQ and ML Benchmarks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.110397Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:036ffada15ae44b59f025b3fe216a1109eacc7d4a8d34bcacc2479a708fb85f9","observation_id":"32835267-c04d-4a63-b5ff-2251fa531963","resolution":{"observed_at":"2026-08-12T19:12:55.110397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.758359Z","title":"Sociolinguistically Driven Approaches for Just Natural Language Processing","venue":null,"work_id":"41df8ee2-141f-4f19-974f-bed588e70bd4","year":2021},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.114777Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:4372fda81c433b035c3381c41ee6d1ce6fa848ff1206a9efa2974b8638d30049","observation_id":"add1c81a-962e-4128-9e3f-b6108121cbff","resolution":{"observed_at":"2026-08-12T19:12:55.763447Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.743550Z","title":"Language (technology) is power: A critical survey of ‘bias’ in nlp","venue":null,"work_id":"9524f294-ca85-4958-88b7-f9ddd3fcc6e2","year":2020},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.120069Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:831bec1c0d62ad1e068fe6c7afddfeb71c613d95164a170366f905e06c1c4e57","observation_id":"101e6a60-6366-44ab-8b14-386973d310db","resolution":{"observed_at":"2026-08-12T19:12:55.748677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.728774Z","title":"Stereotyp- ing norwegian salmon: An inventory of pitfalls in fairness benchmark datasets","venue":null,"work_id":"eb2b8de8-ab6e-401f-a2c5-f4b05a9d855f","year":2021},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.125557Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:2ccc3716a0c8898fd3bf4c3fde246f91c15decee10eb1192fac788d29249aa8c","observation_id":"1bc8f043-44c9-4829-a536-0649cd83b47f","resolution":{"observed_at":"2026-08-12T19:12:55.733739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.04053","last_updated":"2023-08-30T18:41:01Z","snapshot_observed_at":"2026-08-19T17:33:29.206707Z","submitted_at":"2022-02-08T18:36:52Z","title":"DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.04053","snapshot_observed_at":"2026-08-12T19:12:55.131216Z","title":"DALL-Eval: Probing the Reasoning Skills and Social Biases of Text-to-Image Generation Models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.131216Z"},"links":{"cited_paper":"/paper/2202.04053","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:51a01e3de0f6cb06c9019f866c665a171cbc4057fffc5950dafba9729a0466f1","observation_id":"38afce52-aa80-413e-8b11-abab2ad3eda6","resolution":{"observed_at":"2026-08-12T19:12:55.131216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.136919Z","title":"Feder Cooper, Ellen Abrams, and NA NA","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.136919Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:fc7ed0af09af593f73bd99e4b76be2763269a9bef2576129ef0ca1f065a69b16","observation_id":"986f4050-05e4-4824-b4b9-a11795e3d20b","resolution":{"observed_at":"2026-08-12T19:12:55.136919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06477","last_updated":"2023-12-03T03:09:16Z","snapshot_observed_at":"2026-08-16T14:44:12.147413Z","submitted_at":"2023-11-11T04:13:37Z","title":"Report of the 1st Workshop on Generative AI and Law","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06477","snapshot_observed_at":"2026-08-12T19:12:55.142295Z","title":"Feder Cooper, Katherine Lee, James Grimmelmann, Daphne Ippolito, Christopher Callison- Burch, Christopher A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.142295Z"},"links":{"cited_paper":"/paper/2311.06477","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:dbdff0d5798bd18ed6d3a738be578784049554fbd7f8e3fcb233c8ce778e7fb8","observation_id":"90b5c918-b4b2-44d1-852e-9292b1a017a6","resolution":{"observed_at":"2026-08-12T19:12:55.142295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.147508Z","title":"Representational harms through the lens of speech act theory","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.147508Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:33b9169e46e6c9a96e3a5af2c7b3192101181c52ff1a8fcdd61ae56dc74dbacb","observation_id":"8467d862-bdb5-40a0-9609-f968ebab6d99","resolution":{"observed_at":"2026-08-12T19:12:55.147508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.702516Z","title":"Construct validity in psychological tests","venue":null,"work_id":"1a0a9313-3367-474e-aa6d-da52e228a8ca","year":1955},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.151808Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:49d6343b6bb6be3e67ce92405ba3aeadec0934c6e59e87d17fa1cb0d03a472b9","observation_id":"75e2a104-d303-4c69-9809-96810cc6ed3b","resolution":{"observed_at":"2026-08-12T19:12:55.707824Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.686617Z","title":"Measurement and fairness","venue":null,"work_id":"0be22fbe-9b29-4b17-8f06-2b7d4e095695","year":2021},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.156530Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:0d363bc14af2a542d2e6edcf9844c336e43d147685faa08ea8c0e383d4f47340","observation_id":"5f8e446b-2916-46e3-947c-dbfd020ed0e0","resolution":{"observed_at":"2026-08-12T19:12:55.692482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06310","last_updated":"2024-07-14T21:17:05Z","snapshot_observed_at":"2026-08-21T13:12:32.218867Z","submitted_at":"2024-01-12T00:43:57Z","title":"ViSAGe: A Global-Scale Analysis of Visual Stereotypes in Text-to-Image Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06310","snapshot_observed_at":"2026-08-12T19:12:55.160667Z","title":"Reddy, and Sunipa Dev","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.160667Z"},"links":{"cited_paper":"/paper/2401.06310","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:fbdedca4011bd53bf4075a67af5bcaacf5dceca00844239b5f37579240fc0fff","observation_id":"d8dd5ef3-fb54-4dd9-a304-c9c7657961a8","resolution":{"observed_at":"2026-08-12T19:12:55.160667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.672404Z","title":"Hate speech in public discourse: A pessimistic defense of counterspeech","venue":null,"work_id":"e4535211-398a-4e04-b2d3-df97fc1ec02b","year":2017},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.165191Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:2af436a87fb48a435079d1abbab5aa4290a26e851b2eb1aa083da7bdc01d025c","observation_id":"648da852-aae6-476c-b7ce-e6222b1123c2","resolution":{"observed_at":"2026-08-12T19:12:55.676896Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.657643Z","title":"Vera Liao, Alexandra Olteanu, and Ziang Xiao","venue":null,"work_id":"03b7c1ac-9e90-4906-8f2f-05b11caa28f1","year":2024},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.169333Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:9952acbbf90381a3a55e94fb2c65aabbe966a998e24b640c53e2944da8169c55","observation_id":"22e50b36-cf29-499a-b48c-59a449cd2b04","resolution":{"observed_at":"2026-08-12T19:12:55.662480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.642540Z","title":"Validity and washback in language testing","venue":null,"work_id":"94393789-e3e9-4e2c-a857-c6e4f47cedee","year":1996},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.173553Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:ce78de01b6fef5217ae9836eebba82d26dc224c9e07142facc3b3d846c18db76","observation_id":"2f209101-db34-418b-a817-9002c6ae1989","resolution":{"observed_at":"2026-08-12T19:12:55.647368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.628061Z","title":"Privacy is an essentially contested concept: a multi-dimensional analytic for mapping privacy","venue":null,"work_id":"8f3104f1-99e4-4a5c-a104-eb76b113cbb0","year":2016},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.177711Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:4fe616d9394812fe2bf2481fad3870e5a5b07a96fe52dc1364349f35de1281d1","observation_id":"1fe3b165-cfb7-48bb-8116-eddcb5d620e2","resolution":{"observed_at":"2026-08-12T19:12:55.632568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.613687Z","title":"Mulligan, Joshua A","venue":null,"work_id":"7359c0e3-78a7-4bc1-b938-fef749fad422","year":2019},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.181927Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:a9337c1ddced7e77a23bed5623277aeb4365c28bd4fd2946819529896da57ff5","observation_id":"a6ab75ab-7f1b-430b-8c02-5cd9e255dc96","resolution":{"observed_at":"2026-08-12T19:12:55.618323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.09456","last_updated":"2020-04-20T17:14:33Z","snapshot_observed_at":"2026-08-18T18:59:00.636649Z","submitted_at":"2020-04-20T17:14:33Z","title":"StereoSet: Measuring stereotypical bias in pretrained language models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.09456","snapshot_observed_at":"2026-08-12T19:12:55.185915Z","title":"Stereoset: Measuring stereotypical bias in pretrained language models","venue":null,"work_id":null,"year":2004},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.185915Z"},"links":{"cited_paper":"/paper/2004.09456","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:5384faf036cde70857d04f56e4e7db756ec42fc9676b8cb54471ab0284d494b8","observation_id":"7ede09c7-c4b9-41b5-82bf-25ba3bad6834","resolution":{"observed_at":"2026-08-12T19:12:55.185915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.00133","last_updated":"2020-09-30T22:38:40Z","snapshot_observed_at":"2026-08-16T19:16:42.359587Z","submitted_at":"2020-09-30T22:38:40Z","title":"CrowS-Pairs: A Challenge Dataset for Measuring Social Biases in Masked Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.00133","snapshot_observed_at":"2026-08-12T19:12:55.190662Z","title":"Crows-pairs: A challenge dataset for measuring social biases in masked language models","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.190662Z"},"links":{"cited_paper":"/paper/2010.00133","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:549954da10b188fe157ef7b2b47f4c6b8a31f4dcb41831f5f8fb276f37789635","observation_id":"bf0a1e00-ce68-4a96-b888-926f9f26f5e2","resolution":{"observed_at":"2026-08-12T19:12:55.190662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.597106Z","title":"Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile, 2024","venue":null,"work_id":"1a42b0cd-4a4b-4574-8797-8a6e5868b94f","year":2024},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.195061Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:72876aede69b7180c8d3e30482f75599384da5ad9cf471b734c2894ecb190aea","observation_id":"0736818e-a22b-4792-b0b0-ccbbdf97d5c3","resolution":{"observed_at":"2026-08-12T19:12:55.602679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.03286","last_updated":"2022-02-07T15:22:17Z","snapshot_observed_at":"2026-08-15T19:46:54.909325Z","submitted_at":"2022-02-07T15:22:17Z","title":"Red Teaming Language Models with Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.03286","snapshot_observed_at":"2026-08-12T19:12:55.199612Z","title":"Red Teaming Language Models with Language Models, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.199612Z"},"links":{"cited_paper":"/paper/2202.03286","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:06e268b848479974b03b5b9131574fe16a54dc1b7a119074549ad78ca072156d","observation_id":"fecc7a41-e79f-4c13-bd96-80764b4160f1","resolution":{"observed_at":"2026-08-12T19:12:55.199612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16379","last_updated":"2023-12-29T05:42:07Z","snapshot_observed_at":"2026-08-17T22:11:37.761830Z","submitted_at":"2023-10-25T05:38:38Z","title":"Evaluating General-Purpose AI with Psychometrics","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16379","snapshot_observed_at":"2026-08-12T19:12:55.204761Z","title":"Evaluating genera-purpose AI with psychometrics","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.204761Z"},"links":{"cited_paper":"/paper/2310.16379","citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:a8c7a45240c4f234024aa6605985ad76e4ae066e5a5e0ff93e0297593909429d","observation_id":"fba9a1d0-413e-4320-abf7-c05e07483ebd","resolution":{"observed_at":"2026-08-12T19:12:55.204761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T19:12:55.581393Z","title":"The nature and origins of mass opinion","venue":null,"work_id":"fdb87e49-592a-4597-a1a2-31825e721019","year":1992},"citing_paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T19:12:55.209578Z"},"links":{"citing_paper":"/paper/2411.10939"},"observation_digest":"sha256:9adf2c3ae410db4f5e4788851c0895c745c926e88b72b3e497740a5b726de9a4","observation_id":"51eb7304-e85e-4cce-940b-4126f761dd64","resolution":{"observed_at":"2026-08-12T19:12:55.587253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.10939","last_updated":"2024-11-17T02:35:30Z","latest_version":1,"primary_category":"cs.CY","snapshot_observed_at":"2026-08-18T18:59:35.955040Z","submitted_at":"2024-11-17T02:35:30Z","title":"Evaluating Generative AI Systems is a Social Science Measurement Challenge"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":1,"verified_fuzzy":12},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 11 inbound Pith citation observations for arXiv:2411.10939."}