{"as_of":"2026-08-11T13:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:68ff41ac9758f269922396c3f8045ddd7bec57433318859c2ab79fc1e51bdbbc","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T11:46:09.108129Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T04:16:26.175093Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T04:18:56.704885Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"cited_work":{"arxiv_id":"2601.18061","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.18061","snapshot_observed_at":"2026-08-05T04:18:56.704885Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","venue":"cs.AI","work_id":"8e98e813-1ae6-4d24-b477-4fb083104e52","year":2026},"citing_paper":{"arxiv_id":"2608.00794","last_updated":"2026-08-05T05:37:25Z","snapshot_observed_at":"2026-08-08T23:10:43.126647Z","submitted_at":"2026-08-01T17:50:12Z","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T04:18:54.595022Z"},"links":{"cited_paper":"/paper/2601.18061","citing_paper":"/paper/2608.00794"},"observation_digest":"sha256:94af35e09ae2322e0f224bdff4dd451f6b9a32a39fc82b0d511bf6e4009d85b3","observation_id":"cdf42a1e-2a34-4555-8b2b-3949d90c5286","resolution":{"observed_at":"2026-08-05T04:18:56.830854Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.18061","snapshot_observed_at":"2026-08-06T04:16:26.175093Z","title":"Jafari, P","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00794","last_updated":"2026-08-05T05:37:25Z","snapshot_observed_at":"2026-08-08T23:10:43.126647Z","submitted_at":"2026-08-01T17:50:12Z","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T04:16:26.175093Z"},"links":{"cited_paper":"/paper/2601.18061","citing_paper":"/paper/2608.00794"},"observation_digest":"sha256:af22796330deffedaec6e59cff087b1c618733294fc56fd91342920ceddb64c0","observation_id":"e1bc1de0-ecac-4a49-97c8-886f6aa76f8b","resolution":{"observed_at":"2026-08-06T04:16:26.175093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2601.18061/citation-record","integrity":"/paper/2601.18061/integrity","json":"/paper/2601.18061/citation-record.json","paper":"/paper/2601.18061"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Clinician-Rated Severity of Nonsuicidal Self-Injury","venue":null,"work_id":"e6ff08c3-df87-40e0-869a-ccb5d7da797e","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:37fd23451042f1c468e2e5fcd894f26f7bbe4cfa88bffaf96ae35b251b00c226","observation_id":"89705895-d623-41a1-9534-e24224803783","resolution":{"observed_at":"2026-05-16T11:47:49.788739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DSM-5 Clinician-Rated Dimensions of Psychosis Symptom Severity","venue":null,"work_id":"43443a03-329c-4557-87b9-8bebd6699dc1","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:742cc894ab83a2e5c3cf95d2c803ffba917af857b38d620536dd84aa6f10c18c","observation_id":"0adde6b6-9e29-4ee8-8ce8-dee16e65a5ba","resolution":{"observed_at":"2026-05-16T11:47:49.802673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"DICES Dataset: Diversity in Conversational AI Evaluation for Safety.Advances in Neural Information Processing Systems, 36:53330–53342","venue":null,"work_id":"a23dc6ba-7912-4251-9d0e-a40fc29f0dd9","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:980374a943c62a46c7a0a7f0bf258fdad3347133a64e3bd365be8d3caaec3f3a","observation_id":"1e4bec4c-d8b2-4dc8-90c3-a88efc156446","resolution":{"observed_at":"2026-05-16T11:47:49.797026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Truth Is a Lie: Crowd Truth and the Seven Myths of Human Annotation.AI Magazine, 36(1):15–24","venue":null,"work_id":"7385128e-2705-4765-b6c7-4461bb60ff7e","year":2015},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:2c2f3cb5919dffb56a04fec5b99bd186e21894006acf0853a9d73b673300576c","observation_id":"05f8dcd9-93a8-4a4b-b3a1-d00b890fe36c","resolution":{"observed_at":"2026-05-16T11:47:49.799885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Crowd Truth: Harnessing Disagreement in Crowdsourcing a Relation Extraction Gold Standard","venue":null,"work_id":"b65099c0-4dcf-4c62-a2c9-6f46a21daafc","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:59787f117c3862e46e33e0e85312b6ebf0e9a996cae79fe2f51049e05bace3d0","observation_id":"e9922f6e-93e3-40d3-b196-7528be7f4d82","resolution":{"observed_at":"2026-05-16T11:47:49.777957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bowman, Zac Hatfield-Dodds, Ben Mann, Dario Amodei, Nicholas Joseph, Sam McCandlish, Tom Brown, and Jared Kaplan","venue":null,"work_id":"a37f0da3-58a7-4640-bd28-69ee2c035b9b","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:1d623352c437c10f379c3c39d91f08f0b90d8dc59e2e5f4f1f44aecfc710a411","observation_id":"4e6e9d3c-fcdb-419a-8271-a99241af5c35","resolution":{"observed_at":"2026-05-16T11:47:49.775559Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bech.Clinical Psychometrics","venue":null,"work_id":"5e7413ea-80cf-49ec-9d3a-02acc9da7759","year":2012},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:87d56b676251bde8b83ff3c37242c48e6c598124ef420972153160415dc2ab60","observation_id":"f2e61751-f922-45ee-9b08-979a55795f70","resolution":{"observed_at":"2026-05-16T11:47:49.794378Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2a00a35d-6ab7-4d7f-acaf-9c433a471147","year":1974},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:349981f41205c7a6b4839cfde3a9de1da8b52890a49775031a8dc29adabbcf29","observation_id":"0d376012-b765-4928-8354-102339370c21","resolution":{"observed_at":"2026-05-16T11:47:49.791417Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Consensus report of the apa work group on neuroimaging markers of psychiatric disorders.Am Psychiatr Assoc","venue":null,"work_id":"79bf7702-304c-4266-acca-37899768becc","year":2012},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:3502f3ef95bf104eb27f871cc4f9b894a60433fcb3a6effc9f5336d4e5245f21","observation_id":"55997058-543a-4706-a361-f2c18a69573d","resolution":{"observed_at":"2026-05-16T11:47:49.771018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Using Thematic Analysis in Psychology.Qualitative Research in Psychology, 3(2):77–101","venue":null,"work_id":"81ea3b20-361a-4282-a933-8f86c4d15eb3","year":2006},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:9e1f2e5526e4e1bed48a4b61ae2b9d1e34393857ff1e46e79c9e0dbd42bfb18b","observation_id":"940f40b0-d336-49c5-9f1a-d098a446facc","resolution":{"observed_at":"2026-05-16T11:47:49.785997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minton, Abigail Lott, and Jinho D","venue":null,"work_id":"6b45cebd-2604-4fe6-bafa-2aacf02c8b65","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:186e9af5536b6fcd22771f202eaf9a5d948c424c17bbe0241c375bb93bafca0d","observation_id":"e77b00b4-e418-4355-9ed3-960f0ca0f5cd","resolution":{"observed_at":"2026-05-16T11:47:49.780735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Open Problems and Fundamental Limitations of Reinforcement Learning from Human Feedback","venue":null,"work_id":"e34a7753-54f8-40bf-8463-fae21727b9d2","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:b635d6f728d86dc473d987bd18cbd23f61c45740fa06c5f47023c3dbd435e380","observation_id":"5c35b276-3251-4a8b-8e82-3ed483f4f883","resolution":{"observed_at":"2026-05-16T11:47:49.692465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T18:41:20.050987Z","title":"How people use chatgpt","venue":null,"work_id":"141200a0-c50b-473e-ad07-f653b2448e00","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:dec7c5c9c3537126031e602e035b997260d36913e739b3e9a3585e48a0e19418","observation_id":"d5270066-a82f-4590-87f7-3912b967b006","resolution":{"observed_at":"2026-05-16T11:47:49.641532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Predicting Depression via Social Media.International AAAI Conference on Web and Social Media, 7(1):128–137","venue":null,"work_id":"667ff3cd-db24-4b31-8ce5-df1fb0894d00","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:1d7577094ce86e62a9692f969b4d04ae49d89e9a52ee67f8c7f7f1a94c24052c","observation_id":"e7ba2d63-d3bf-49e4-ad8a-1ce5804a2c25","resolution":{"observed_at":"2026-05-16T11:47:49.653085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep Reinforcement Learning from Human Preferences","venue":null,"work_id":"895f578f-9a2e-45c6-a5e1-82d69cb52462","year":2017},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e813f3ec41d23c991d09a9a9417ad03278a9688d7952d9c5b652504dde58cf8c","observation_id":"28ab72d8-fa4f-4ac1-9997-2d9a277ac1fd","resolution":{"observed_at":"2026-05-16T11:47:49.650228Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cicchetti","venue":null,"work_id":"1a69a631-1527-4f90-8aea-a5700dd30398","year":1994},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:05ed25660f38a1a554ac0c9faad975b09c5768862201faa228b8aae6feb30b91","observation_id":"4e60b9a8-85f4-4021-bc97-fbed1d805f31","resolution":{"observed_at":"2026-05-16T11:47:49.661356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hashimoto","venue":null,"work_id":"39302b36-f027-401e-b599-cc5fde23950f","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:553bfeae1364e79f49ed57e434497e5480f7f2e280f58a43e29ba2191e6edccf","observation_id":"5f765e66-41fa-4f04-b2a7-08f9b4020713","resolution":{"observed_at":"2026-05-16T11:47:49.658470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Diagnostic and statistical manual of mental disorders.Am Psychiatric Assoc, 21(21):591–643","venue":null,"work_id":"372c6dca-23fa-49f1-824d-a73b6b03c833","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:71dd3524e7aa127bba172dcb202aa46453fbd2a2adfd689f36c1efaea2cba46a","observation_id":"793adfb1-7bd6-409d-8b4c-9fad441839d2","resolution":{"observed_at":"2026-05-16T11:47:49.735587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1037/t03974-000","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"PsycTESTS Dataset","work_id":"dfdaaa7b-eb64-4977-a81f-1543801394f2","year":1994},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ed73d6fba843f1afc26a247b6e00b15cd19402c9f6f4908596a3b671443230e2","observation_id":"792b627f-be36-4f30-ae29-b1e0dac0468b","resolution":{"observed_at":"2026-05-16T11:47:49.282324Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"8409be2e-e5a6-4ff4-a824-1f496125e417","year":2017},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:b6fe6452c349eaa1a35bdb1c69cceff8eb0a3deca4cb45f2c367762d2005df22","observation_id":"212b9a9d-18d0-4aca-8802-7d3bbb48e4a0","resolution":{"observed_at":"2026-05-16T11:47:49.624465Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Can AI relate: Testing large language model response for mental health support","venue":null,"work_id":"b45d0962-085d-4a47-92f6-4cfe8e887583","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ef7307f5278dfa911a5cff261221d5beb301c0ef67ade2797adab3d31bceb111","observation_id":"598aa3da-0aeb-4af6-9d2c-1e2b4133470f","resolution":{"observed_at":"2026-05-16T11:47:49.633249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Impact of preference noise on the alignment performance of generative language models","venue":null,"work_id":"a5314b14-c762-484a-928d-9e588dda221b","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:c93823567be753150dc05ed6d696e6e7a6cec521612f0149f7e8aa483a024b57","observation_id":"b3d75980-8685-4f1f-b133-7e00f5295efc","resolution":{"observed_at":"2026-05-16T11:47:49.704854Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blind spots and biases: Exploring the role of annotator cognitive biases in NLP","venue":null,"work_id":"3fa2f1e1-e20a-46f8-adb4-1ed1dc25ac3f","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:f9da9e5479c13aa4940d316132e66df2acaae366f96f0f5e327d157b2383300d","observation_id":"942edd00-ba09-4dff-8d65-bcc7949576ee","resolution":{"observed_at":"2026-05-16T11:47:49.619087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Goodman, Lawrence H","venue":null,"work_id":"382261ca-ed4b-40e6-8de6-dcb451fcb9b9","year":1989},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d65f5261152289613269367d3fadcea5e34406a965feed0bd0353d5264079f1f","observation_id":"26db7dd9-5be6-4967-b51f-cec76322a12d","resolution":{"observed_at":"2026-05-16T11:47:49.639057Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gordon, Michelle S","venue":null,"work_id":"d710ccea-f18c-42bc-93aa-db7e27166872","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:83fbf20fa121d13cad61263e6d846de8480fa53e225ee9ce630f53879d97b61b","observation_id":"027577f7-8577-4fcc-9c3a-e8e95fbf536f","resolution":{"observed_at":"2026-05-16T11:47:49.723443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Risks from language models for automated mental healthcare: Ethics and structure for implementation","venue":null,"work_id":"681a00c4-2c79-461c-9e86-51da71d16c01","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:69a374305435d5f26b2c4db7a2a35df199cce64fdabacb7a1c62b3b7c789edef","observation_id":"38e14316-48b9-4210-b0e8-dcd5e5fc5685","resolution":{"observed_at":"2026-05-16T11:47:49.768121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Human Feedback is not Gold Standard","venue":null,"work_id":"bd0de5ef-9311-44c8-a268-035e19496993","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:ac5264b12b24674b54b6f2fdec0c7190f78afcb82838bd80a728fcad9b62abd6","observation_id":"f73d4089-4f83-4d33-ab31-a800c9d1bae7","resolution":{"observed_at":"2026-05-16T11:47:49.689018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How LLM counselors violate ethical standards in mental health practice: A practitioner-informed framework","venue":null,"work_id":"fd6cd067-e08e-47eb-ab74-8527bda86fe8","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:dd824472e85b84c2856a12fd664cdb6cf36e72099293d66ae07276d6da300bda","observation_id":"98fac7e5-2c51-4c5d-a5ba-31b4f65ba2b9","resolution":{"observed_at":"2026-05-16T11:47:49.783275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","venue":null,"work_id":"c496bd12-cd9a-4782-9e1e-a150b376fe53","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d5f8b6563e8a5cace3ce71fbb8d80df48718235924760b9a37042d2b90315502","observation_id":"33cf71bc-79df-4819-ada5-1bedf766dccc","resolution":{"observed_at":"2026-05-16T11:47:49.670029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fc2dfa18-14ad-4c99-871b-dee40ecb0a2c","year":2018},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:53260aa40b742fa0717704335ede23acdd5fdbe9e59eafe9fd175a2ea491e098","observation_id":"428a34c9-6494-48b3-bc1d-ece9f3437232","resolution":{"observed_at":"2026-05-16T11:47:49.701487Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4fbeb063-1e06-4fe1-bfec-88fab3162a40","year":1975},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:985abc7b81c44bd0012612c8e888c16d154a26940e5f3b5b5b54c0d9d65ca0d2","observation_id":"f48788c3-4db0-498f-addc-942c10481929","resolution":{"observed_at":"2026-05-16T11:47:49.683843Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reliability in Content Analysis: Some Common Misconceptions and Recommendations.Human Communication Research, 30(3):411–433","venue":null,"work_id":"89ae0edf-6723-4634-968e-ff2fb3d5d819","year":2004},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:6ecaa24439a35a0e0d03f90425d24de687ca23a8a7af2209ba768aaa8a4e9a7a","observation_id":"65cf5188-ab62-47c4-a72b-a9555075c5e4","resolution":{"observed_at":"2026-05-16T11:47:49.655706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Kunstman, Aaron Lulla, Monika Drummond Roots, Manu Sharma, Aryan Shrivastava, Nina Vasan, and Colleen Waickman","venue":null,"work_id":"ffdfeb0c-a929-4e03-9d9b-32285523113f","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:f386b10e976c1dec2eea38ba1106949327c22040e5c8e6f9b5bb8390d5a4cfba","observation_id":"4f915837-2483-460e-bffe-798b41f4a387","resolution":{"observed_at":"2026-05-16T11:47:49.698419Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dcc5fa87-4958-4fe0-9dda-c5f2fc1fba27","year":2016},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:4db431432a4bde698d6b2bafcac23e4b07d5b7f769e745d05ef87efb2e32e142","observation_id":"257c5123-393b-483e-abc0-32495fe4c450","resolution":{"observed_at":"2026-05-16T11:47:49.636158Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hashimoto","venue":null,"work_id":"885dc18a-4cbf-4307-b726-a7adda854b61","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:7d8c6d32846b397c703354f2c51188f775ca0cfce0960eedb11cfe9b562465c7","observation_id":"8d28453c-fef4-4410-a4be-afee8f1fde6f","resolution":{"observed_at":"2026-05-16T11:47:49.664401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bunyi, Adam C","venue":null,"work_id":"00cb89c0-3994-4bab-a3a1-a9319ae157b4","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:519de5e8052f48e21b4cbb74916996b241f5b6ff7eca04a516fedfbd41f1b897","observation_id":"4731ff42-453d-448c-b60a-aea94481133b","resolution":{"observed_at":"2026-05-16T11:47:49.627248Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sample Size Considerations for Fine-Tuning Large Language Models for Named Entity Recognition Tasks: Methodological Study.Journal of Medical Internet Research AI, 3:e52095","venue":null,"work_id":"a6afb227-dda6-4f17-af4a-16fd0d26aaf3","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:6fd67fd90474516cacf6443ccf9c1ae76ac47c38465f88161c67ed443893565e","observation_id":"c6897c9c-874c-422a-8468-6affd4168880","resolution":{"observed_at":"2026-05-16T11:47:49.681405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A diagnostic meta-analysis of the patient health questionnaire-9 (phq-9) algorithm scoring method as a screen for depression.General hospital psychiatry, 37(1):67–75","venue":null,"work_id":"340c0d9e-815d-4119-9bfe-c01fb7564d91","year":2015},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:b912a91fa9809e4b0f5e28296f2312015e58b59f150ecdee9c07005050a21be5","observation_id":"bfae3754-10ce-452d-9296-72d1725222a7","resolution":{"observed_at":"2026-05-16T11:47:49.616428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"McGraw and S","venue":null,"work_id":"b118cb7e-a290-4b97-859d-e41c7eeba396","year":1996},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:2084b6886219dcd9f3826f5565acaf5b42d276b6c980dfc64ef2362d4ab05004","observation_id":"fb3125eb-1f42-49fc-884e-8dea0af2abc7","resolution":{"observed_at":"2026-05-16T11:47:49.678285Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ong, and Nick Haber","venue":null,"work_id":"f20be75a-49e6-4ba8-88de-ce63dc786d75","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:32f4ee3b2a4a25720c2bcb8f4e8abe8b6b69440f408b8b8cfe7fb14a3429d8a7","observation_id":"fc5bb067-0f13-4b44-9a31-6fe731312cf2","resolution":{"observed_at":"2026-05-16T11:47:49.686452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Moyers, Lauren N","venue":null,"work_id":"3d97fe22-52cd-45c3-aebf-94220df2f5aa","year":2016},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:35849aabc37f6343405d0a1c612b3936981f77979aaa06304399a9fadff0151c","observation_id":"4d351919-2dd3-4888-99f0-6bac4c34a7b2","resolution":{"observed_at":"2026-05-16T11:47:49.647305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Department of Veterans Affairs","venue":null,"work_id":"a810a70e-7f23-4279-ab01-6318b41d7149","year":2015},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:a36e4f6747866a3ad794e98a219ffe73af944ed59544cbd29ff6ff8dcd92f49b","observation_id":"59346976-8a19-4392-a242-d53521eaf689","resolution":{"observed_at":"2026-05-16T11:47:49.739563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NICHQ Vanderbilt Assessment Scales","venue":null,"work_id":"1aeef63a-3e80-430d-97a4-e45b083ca187","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:bf16043d94001ca1f78919671b6fdf908ec39b2402b239568b75e087a0cc3dec","observation_id":"7b07b7ee-fa44-4bde-be7d-1099d2d3f527","resolution":{"observed_at":"2026-05-16T11:47:49.720304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Depression","venue":null,"work_id":"5d86dd17-c53a-4228-9eb4-d022b4260040","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:0967d41aa6275547e6164952ba39f890e2904a50c43a754bcba3ee50a96a88e1","observation_id":"98ca2986-71ec-40ee-8239-f70dbe2cf385","resolution":{"observed_at":"2026-05-16T11:47:49.695200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enhancing mental health with artificial intelligence: Current trends and future prospects.Journal of medicine, surgery, and public health, 3:100099","venue":null,"work_id":"06207350-b486-47e5-a61f-5cd615bb273d","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d6785aff2eed38dd8b982caa7d1c5a6d6d4a4be6877b8f681786bcb24f6fa829","observation_id":"fff6dc70-4901-4d49-8b6a-d778979de163","resolution":{"observed_at":"2026-05-16T11:47:49.732269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Christiano, Jan Leike, and Ryan Lowe","venue":null,"work_id":"9a523403-0c70-44fc-97a0-8ac4dc4ce6f1","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:31ddaab203619cede5ff60a6c87b86ce4fa74f223991b6fc569e828fd4aea645","observation_id":"00113e04-ee5f-4e6f-ab52-b7563cfa53ba","resolution":{"observed_at":"2026-05-16T11:47:49.675515Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Inherent Disagreements in Human Textual Inferences.Transactions of the Association for Computational Linguistics, 7:677–694","venue":null,"work_id":"ddede9b7-90c4-4917-9133-8e3ebfc05665","year":2019},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e9597051de70670ec1de7be1953b42ea9610d813bac284a8b95c0d514f6fa642","observation_id":"d4b3fcbe-96fb-417e-9b84-6d5f98ee1736","resolution":{"observed_at":"2026-05-16T11:47:49.644545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Red Teaming Language Models with Language Models","venue":null,"work_id":"dd6714c1-9d60-495a-b8d4-dd345ba3bde4","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e02c4c47028d6937a5c0ab3f1354c491c30dcd3b2c401b0ba0759afac40d8a27","observation_id":"70a2be2c-e2e5-49c6-b176-fb9157ec16cd","resolution":{"observed_at":"2026-05-16T11:47:49.672570Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Posner, D","venue":null,"work_id":"96b774c0-2e0e-482b-a368-c0d140a8a065","year":2010},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:3fb404967490c1231f311ef0d2a2f27b689e7a61a5f3d5f376a4043d9426dce9","observation_id":"7ad7fdb4-d134-4f39-a202-dfbe0a93e440","resolution":{"observed_at":"2026-05-16T11:47:49.630361Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prochaska, Erin A","venue":null,"work_id":"b412c1ab-a728-4f3b-aaf4-a0399bc31aaf","year":2021},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:205873662ac79f072b278a8e1e457a89bd4e900e16aeeff0513fbe071f722a3c","observation_id":"ba9f5582-3a8e-4bf6-ba85-08a669a79415","resolution":{"observed_at":"2026-05-16T11:47:49.729386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Manning, and Chelsea Finn","venue":null,"work_id":"946729f8-023f-4e95-82f4-639378f2c398","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:890be3b6c2fc80bd40c4b37361588d927a954efd8392ce2aeb8fb0cd745b49f9","observation_id":"e65ef78c-0a0e-4f46-a5cf-4677acdf2c8a","resolution":{"observed_at":"2026-05-16T11:47:49.621831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Regier, William E","venue":null,"work_id":"0dee9dcf-a484-4bc4-9ebd-9eae6d209654","year":2013},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:7b4820dab2d6c7a839b6c7ad2fb6e7812153906a882eb128ce7a6aabd9d6f813","observation_id":"b76de69a-9338-4b1e-b092-3f5d05798da4","resolution":{"observed_at":"2026-05-16T11:47:49.742505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Large language models as mental health resources: Patterns of use in the united states","venue":null,"work_id":"d7640ba9-5347-4bba-8cb4-f7d0c1d908e2","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:9ee68f633f105f5b1b9f5d5d922787bf175a8bc62d1b1eb7b11a6bc229ecd522","observation_id":"04ef73ce-2388-4902-93be-9bdb52f0d312","resolution":{"observed_at":"2026-05-16T11:47:49.711024Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b96282ef-4f8d-47b2-9f8a-9be04c7e7f1c","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:f94454d4d81fd01cd7fb47fb7451f55e8f02777f8bc1d1a0c08a95a19da9554f","observation_id":"98307c84-6c72-4814-8ecf-597ffa8652cd","resolution":{"observed_at":"2026-05-16T11:47:49.745311Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lin, Adam S","venue":null,"work_id":"222fde58-8e15-4631-89d7-9329d86bd009","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:63ee501fadc6817ea6259da5ad0e0e2c8c99b88d518bfa041533159110578b12","observation_id":"50d4e86b-a636-44a0-a658-f91e3e379ab1","resolution":{"observed_at":"2026-05-16T11:47:49.762597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Computational Approach to Understanding Empathy Expressed in Text-Based Mental Health Support","venue":null,"work_id":"9a975741-d823-4402-bf28-5d4f690c4eb9","year":2020},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:263a2d2c9310588539fe4690f1cd63463997ddf2237caaffc9ed25e097a7cf5b","observation_id":"db133a62-4faa-41a7-82f7-d4b4065e615f","resolution":{"observed_at":"2026-05-16T11:47:49.667265Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"36d09e2e-2cf5-468f-bdab-3bfc974df8b9","year":1979},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:dce9657505ab4c6fa3c4a2b3754eef6ff2a5b59eb283bc4a13948fd451b3426b","observation_id":"752050e8-e4d8-4045-8ed3-b4cbd6a81428","resolution":{"observed_at":"2026-05-16T11:47:49.707700Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Clinical Practice Guidelines on using artificial intelligence and gadgets for mental health and well-being.Indian Journal of Psychiatry, 66(Suppl 2):S414–S419","venue":null,"work_id":"1837ffab-f778-42c0-9b45-df7e4bb25616","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d708f34060b2bf406a9849e7e3beb39de770d4b4e54b682d5fc2ca2d8cdefeda","observation_id":"54393a32-3d9c-4bc0-aca0-aa974f33a79c","resolution":{"observed_at":"2026-05-16T11:47:49.726521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pfohl, Heather Cole-Lewis, Darlene Neal, Qazi Mamunur Rashid, Mike Schaekermann, Amy Wang, Dev Dash, Jonathan H","venue":null,"work_id":"c2a690a2-cf10-44fe-bfab-7701c545ced3","year":2025},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:013b50b8bcee970949a66490200433777a857fbf51c6284744cf5d64d0c1a5c8","observation_id":"6ab45c2b-2947-46ac-9713-ec7a1e4f1c37","resolution":{"observed_at":"2026-05-16T11:47:49.757209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ziegler, Ryan Lowe, Chelsea Voss, Alec Radford, Dario Amodei, and Paul Christiano","venue":null,"work_id":"7a5c3951-bd71-4bb6-90c2-71c3b7a890d4","year":2022},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:d42723a5836ccb13d91b682275e0074534201e65201ad0043ae398964c13508c","observation_id":"f99fe2f6-d891-48eb-958a-008a368e67a1","resolution":{"observed_at":"2026-05-16T11:47:49.717393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A Practical Guide to Fine-Tuning Language Models with Limited Data","venue":null,"work_id":"adb9faa6-44c2-45c9-805e-cf83b513e846","year":2024},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:46f90b7e6b3ce3fc49b075400e512f4234392786f50aa5a0acfc598d847bafd8","observation_id":"b4f3a450-f6f0-4bea-ae9a-071ef5658955","resolution":{"observed_at":"2026-05-16T11:47:49.714235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lukoff, Keith Nuechterlein, R","venue":null,"work_id":"16becbce-aaf5-4529-bab9-0e2c02495339","year":1993},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:b8f7eb3f26173344ef90de251ed2d5060a42cb536779135031ae0c071eb66d66","observation_id":"d00a5e5b-2d99-4f8a-987b-5806e4dab1c6","resolution":{"observed_at":"2026-05-16T11:47:49.754158Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wang, Patricia Berglund, Mark Olfson, Harold A","venue":null,"work_id":"d2f89104-1e1c-40f8-b0dc-472a17c14e2f","year":2005},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:8090ccf05003b97f7df170d878224fd1bddb890320c8e3722ca88d8d5a73377b","observation_id":"0d9e7534-5b1d-422a-a024-16efb4896549","resolution":{"observed_at":"2026-05-16T11:47:49.759924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b665ce9f-dc45-4566-9d0e-0512d7b44069","year":1978},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:1350ba4ca7d925aa93bcec5af9d857f7a30e60c94f2e54848c357d163907d630","observation_id":"750f9970-3b04-4c32-a787-ab37a9b73f95","resolution":{"observed_at":"2026-05-16T11:47:49.765353Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":"525115b0-e513-4c25-86ee-1fee1ce8da84","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:b885b5c1adfc3d00feee1a34518019d106b843312d2f3afc4017a80390f4a2c8","observation_id":"e29dff42-8874-474f-8a6e-480b1e3c836b","resolution":{"observed_at":"2026-05-16T11:47:49.748254Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cold plunges cure psychosis—stop your medication","venue":null,"work_id":"fb3b4b56-8112-4b54-be47-51be4efd2540","year":2023},"citing_paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-16T11:46:09.108129Z"},"links":{"citing_paper":"/paper/2601.18061"},"observation_digest":"sha256:e841eeb08f4208395f097cac171eefb6bdc0cf1af51dfa2271eb0cbdda1c1806","observation_id":"6c6a7069-3c34-4aa4-80cd-17125693c22f","resolution":{"observed_at":"2026-05-16T11:47:49.751327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2601.18061","last_updated":"2026-05-08T23:59:07Z","latest_version":3,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-05T01:59:43.835531Z","submitted_at":"2026-01-26T01:31:25Z","title":"Expert Evaluation and the Limits of Human Feedback in Mental Health AI Safety Testing"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":1,"verified_fuzzy":57},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 2 inbound Pith citation observations for arXiv:2601.18061."}