{"as_of":"2026-08-19T20:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:25a61e852304829f5f601fae5a0ae93cb3f6c4d1e05bbaf0e5692bbda68d64b0","coverage":[{"denominator":118,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T04:51:17.259117Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.11171/citation-record","integrity":"/paper/2608.11171/integrity","json":"/paper/2608.11171/citation-record.json","paper":"/paper/2608.11171"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.630599Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.630599Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:acca948f3d07680657f5891bc29f1fdfbf7546bda1f3cacdcd7f029311d7fb65","observation_id":"8d9ca6f8-ec40-4cd2-9a0e-f359d768f7d3","resolution":{"observed_at":"2026-08-12T04:51:16.630599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.648681Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.648681Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:84f2eeeb837b2fe8d49fd16506148cfce9807166fcb470db3f9df18c3d1e55e7","observation_id":"de088d98-e2f9-4e2d-af73-5700f1053308","resolution":{"observed_at":"2026-08-12T04:51:16.648681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.667018Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.667018Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:b8cb7432ffe29c6666522db5e628a9f788835328f488c6091abad0d67064dde4","observation_id":"bf5ef2da-7092-4ecf-aada-16d42922fd86","resolution":{"observed_at":"2026-08-12T04:51:16.667018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.671317Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.671317Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:c41a761a8b798fedc7f37143eb63d3b619bfe875433bc650fb0170999455ddc0","observation_id":"c011e2b4-367e-4497-9fa6-fa23c58f332f","resolution":{"observed_at":"2026-08-12T04:51:16.671317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.685034Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.685034Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:b4303f6e915b9964141e2afeb329db81655a0667ee76d2dbf57f4a48fd753b1f","observation_id":"c8e8bca1-0a47-4526-81e7-a4ff31fb9f57","resolution":{"observed_at":"2026-08-12T04:51:16.685034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.690286Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.690286Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:556fd43e79499441fdfaa55de9070bdb5a8455d2d753a9e13cbb468255b312d6","observation_id":"baac49ac-a54a-4be1-b053-273459137e61","resolution":{"observed_at":"2026-08-12T04:51:16.690286Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.737630Z","title":"don't forget the teachers","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.737630Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:c8ce9091496371b5b8ae907aecf8f204c5a39922ca89c1ffd9961ab4702e6b8e","observation_id":"c4ffb726-c5f2-42d2-b97f-597a473e3724","resolution":{"observed_at":"2026-08-12T04:51:16.737630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.745842Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.745842Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:07d423dc5481799976ecb9e26977a229bd13054b23168cd913a5af5e1131bbf7","observation_id":"6b16f8c3-3af4-417c-bddc-72aef843c6b5","resolution":{"observed_at":"2026-08-12T04:51:16.745842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.05561","last_updated":"2024-09-30T10:17:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-01-10T22:07:21Z","title":"TrustLLM: Trustworthiness in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.05561","snapshot_observed_at":"2026-08-12T04:51:16.754194Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.754194Z"},"links":{"cited_paper":"/paper/2401.05561","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:c6f9ddada40dc97f913421bd6b15ce9481ecc664d3c2333c07ecb24468dd6442","observation_id":"9870e611-b528-4ab9-9e31-792d137b9324","resolution":{"observed_at":"2026-08-12T04:51:16.754194Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.758736Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.758736Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:7e08452a6f3ebe25ef91af542b35a6a3f226ed2e997d3c1ff36f93b607f91882","observation_id":"46401178-93bd-4429-a47e-4e174c89416b","resolution":{"observed_at":"2026-08-12T04:51:16.758736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.767502Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.767502Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:b4470f25c6d4c418f2d75adda879ed8804af717f4b68a5f2fc610ae0cf4c3269","observation_id":"831c3195-f441-40e5-a3e8-22ca521a3feb","resolution":{"observed_at":"2026-08-12T04:51:16.767502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.771478Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.771478Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:87fc86789f68456bcaff46e380bf9642de739434ed5c3576f680d11d89cc919c","observation_id":"163ffdfc-8028-4fbd-a4a9-71c18d0b6783","resolution":{"observed_at":"2026-08-12T04:51:16.771478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.775652Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.775652Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:59b0f409f1337119035057b5831b1d543b6cd274c4dae985c89d7f99c15a8de0","observation_id":"1a998e40-6912-4942-baef-7f9c23cbe17e","resolution":{"observed_at":"2026-08-12T04:51:16.775652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.811315Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.811315Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:66fcc6cf80082a65e4265913c316526daf1e00cfc0021706efd3885a0650f2ea","observation_id":"645cca4e-9f50-4fb8-93ce-98e5cd658204","resolution":{"observed_at":"2026-08-12T04:51:16.811315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.815380Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.815380Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:cab0c6e074a934f2343540cb215f5168301d30ee4faf53fba17e1bdde65e3276","observation_id":"f74c9f77-3b20-4494-981d-04f072dc2f58","resolution":{"observed_at":"2026-08-12T04:51:16.815380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.827962Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.827962Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:149ad85bef8fc04048cf96d129491509efd988997100781405951d8a2c143a34","observation_id":"b1e16acf-8f1e-42f5-a2a3-a5e70d8948fa","resolution":{"observed_at":"2026-08-12T04:51:16.827962Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.857656Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.857656Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:da43e27448b3408da75ba79784c00445cc7073a1e1e1cb6c57e026ef366b1e82","observation_id":"6041a784-e8cb-42c8-aa8a-ad32cb84e29a","resolution":{"observed_at":"2026-08-12T04:51:16.857656Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.880222Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.880222Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:5337afcb789c553673ef8b5d8b664592de18368c216254ada0e1ad568bff46ce","observation_id":"7cddb846-25b5-4780-9965-db36f7c79c4b","resolution":{"observed_at":"2026-08-12T04:51:16.880222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.11698","last_updated":"2024-02-26T20:41:01Z","snapshot_observed_at":"2026-08-16T15:22:28.807116Z","submitted_at":"2023-06-20T17:24:23Z","title":"DecodingTrust: A Comprehensive Assessment of Trustworthiness in GPT Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.11698","snapshot_observed_at":"2026-08-12T04:51:16.888702Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.888702Z"},"links":{"cited_paper":"/paper/2306.11698","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:1ff4f77547bc2d3ef09627ca669fa8bc2c714e65283c4acaca693849320e1de4","observation_id":"68a32cee-7b52-4259-8f3a-8fe60d1019e6","resolution":{"observed_at":"2026-08-12T04:51:16.888702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.901804Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.901804Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:65f452939e5deee0d1476bc4d55993667c3505605ef0e98f4e4cc507d61cf3be","observation_id":"02bafdd9-53fb-4a9a-a152-1238249fb910","resolution":{"observed_at":"2026-08-12T04:51:16.901804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.919064Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.919064Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:0e2572e53722978e1525e0c8e6d51c1b5128105dc076db3a9afc1060c0b8b14f","observation_id":"3c7f868c-804e-474f-8090-2f9f3b95bf1c","resolution":{"observed_at":"2026-08-12T04:51:16.919064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.923408Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.923408Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:91cd50e66bab2b53312664c6aa1f642375e7331f21a7f7c046de5cbc9270f9ee","observation_id":"0c8f1fc8-d6fe-4850-8274-79fed12bacdf","resolution":{"observed_at":"2026-08-12T04:51:16.923408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.927400Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.927400Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:6e8d2c8eac55f41ad8739a78a6397d7fca05d8a9209dbc7492893c306432f089","observation_id":"db943417-ff45-4671-91b0-ed7794a7416a","resolution":{"observed_at":"2026-08-12T04:51:16.927400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.931692Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.931692Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:e5f5b679819e83b32a9604a2ae3835b1d0dd6eb45c039137c107afd5bceab79f","observation_id":"c3b4fb2d-528f-4e0a-89e8-0382adcebc0b","resolution":{"observed_at":"2026-08-12T04:51:16.931692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.935849Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.935849Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:9c96862e5518b0e7b31338892b026ae1b7b38173bc4c0b9a3fe6e76e10f36869","observation_id":"c479cfcd-9596-4125-8055-ad17f421356b","resolution":{"observed_at":"2026-08-12T04:51:16.935849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.939878Z","title":"2026 , howpublished =","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.939878Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:aae2cac0e1c00d40730ffa04f9ab98998105a850ddc9c144bb47620f02a95c9f","observation_id":"84c9f417-3d2e-4107-bed1-25062866139e","resolution":{"observed_at":"2026-08-12T04:51:16.939878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.944004Z","title":"Human-Centered Explainable AI : Towards a Reflective Sociotechnical Approach","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.944004Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:4919828a19b5aeec62107be8d4a9803c23937d323d6d26e67cdfc27f8c5cc663","observation_id":"26f9553c-e3d4-432a-a788-02e54cf8dc1e","resolution":{"observed_at":"2026-08-12T04:51:16.944004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.948539Z","title":"and Wintersberger, Philipp and Manger, Carina and Hubig, Nina and Savage, Saiph and Weisz, Justin D","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.948539Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:6cec25980c96fee0c737c7da36b16cedc359666f7e0ee5fa7c8c9d2dc47bd508","observation_id":"107975d5-4d78-4da4-8757-05817c127b83","resolution":{"observed_at":"2026-08-12T04:51:16.948539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.03712","last_updated":"2024-12-09T02:14:08Z","snapshot_observed_at":"2026-08-16T13:45:41.906628Z","submitted_at":"2024-06-06T03:15:13Z","title":"A Survey on Medical Large Language Models: Technology, Application, Trustworthiness, and Future Directions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.03712","snapshot_observed_at":"2026-08-12T04:51:16.952670Z","title":"arXiv preprint arXiv:2406.03712 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.952670Z"},"links":{"cited_paper":"/paper/2406.03712","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:eeba3997ce987ca6db4b3c7214a3b7c524a38f9b96fdeb62196465554f00609e","observation_id":"0e2ea7ae-a4f9-46c1-9066-b8b0cf35c35a","resolution":{"observed_at":"2026-08-12T04:51:16.952670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15865","last_updated":"2025-06-02T10:13:24Z","snapshot_observed_at":"2026-08-19T00:10:04.903158Z","submitted_at":"2025-02-21T12:56:15Z","title":"Standard Benchmarks Fail -- Auditing LLM Agents in Finance Must Prioritize Risk","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15865","snapshot_observed_at":"2026-08-12T04:51:16.956798Z","title":"arXiv preprint arXiv:2502.15865 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.956798Z"},"links":{"cited_paper":"/paper/2502.15865","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:34fa24f8ffb90f336bb2eee432f5de8bb86bde494c14c3d70e282b7719415c67","observation_id":"da851883-7758-46cc-820b-29c1860d8b97","resolution":{"observed_at":"2026-08-12T04:51:16.956798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.960742Z","title":"Don't Forget the Teachers","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.960742Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:62f5c221f4fa9e056b9944dc03fdbe47ebe86b4e4a5258f6d8f9ae17837cb0ec","observation_id":"91bc9c9d-60ba-4cbf-9cf8-667c86e04d82","resolution":{"observed_at":"2026-08-12T04:51:16.960742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.965221Z","title":"2025 Silicon Valley Cybersecurity Conference (SVCC) , pages=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.965221Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:1fd77696ee7a6dd8c91532ebb9744188c5aeb77e8c237476f8d57abae1a1483e","observation_id":"913fa84b-6be8-4080-8c9f-6d436179cc43","resolution":{"observed_at":"2026-08-12T04:51:16.965221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.969245Z","title":"2023 , url =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.969245Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:6e35d8db54e84df55e077ba0afa58bb58792b9c18e44a798540ec5b6158f42a3","observation_id":"3bd9c0e2-d80d-419f-a421-d48a1d9b36af","resolution":{"observed_at":"2026-08-12T04:51:16.969245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.973582Z","title":"2024 , url=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.973582Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:73de50c5189b7ca36535e3389b89c24457ad7bdbb49df9060f5a1b7303f7cb33","observation_id":"4cf52372-36ab-4930-b5d2-9427461bb01c","resolution":{"observed_at":"2026-08-12T04:51:16.973582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05374","last_updated":"2024-03-21T00:21:14Z","snapshot_observed_at":"2026-08-02T17:09:20.540788Z","submitted_at":"2023-08-10T06:43:44Z","title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.05374","snapshot_observed_at":"2026-08-12T04:51:16.977677Z","title":"Trustworthy LLM s: A Survey and Guideline for Evaluating Large Language Models' Alignment","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.977677Z"},"links":{"cited_paper":"/paper/2308.05374","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:0264ef31d17dbd80ebc4e68068ff2fdd19f0ea5a3c32a054e80743294c4a663c","observation_id":"428c3343-a31c-4b1a-a90f-2fc70ef11344","resolution":{"observed_at":"2026-08-12T04:51:16.977677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.981843Z","title":"2025 , publisher=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.981843Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:031bb7cdd9078f1789abe3a3478926628e417ec0a54ee8dda77ed0143e440eb9","observation_id":"5efb6439-584c-4411-a97b-fb0b4b0a62ce","resolution":{"observed_at":"2026-08-12T04:51:16.981843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.986141Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.986141Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:dc09270b890bf56c563564061871001837dae0c16963af909915b8b31537e171","observation_id":"498ceeaf-2846-448f-bc20-7fc603cd67ae","resolution":{"observed_at":"2026-08-12T04:51:16.986141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.trustnlp-1.1","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.575165Z","title":"Interpretability Rules: Jointly Bootstrapping a Neural Relation Extractor with an Explanation Decoder","venue":null,"work_id":"bcc043de-1aff-4a67-803b-ec0b573a43c8","year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.990589Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:dde85c43d27571ba75a720f04f73feafd6c72aaf1b3375ead156a9955bce35c1","observation_id":"37a8edc6-c58f-404c-b0a7-82c03f5dcfb3","resolution":{"observed_at":"2026-08-12T04:51:17.579945Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.trustnlp-1.2","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.891481Z","title":"Measuring Biases of Word Embeddings: What Similarity Measures and Descriptive Statistics to Use?","venue":null,"work_id":"0d38967d-d861-4926-b3e0-825e0c27b3b4","year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.994632Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:5c71052a4240bc5a88baa917e2afe4befd857fd5fb1b3594fc2a8783c3c63b3c","observation_id":"fa99828f-2299-4b74-8df4-932c0a7707fd","resolution":{"observed_at":"2026-08-12T04:51:17.897013Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:16.999053Z","title":"and Kiritchenko, Svetlana and Balkir, Esma","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:16.999053Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:f1af7e37ecffc668b301ba7913a1433bafca3a34379099dcedac7bc77bd66c49","observation_id":"0d500bcb-aa3f-47bf-870e-b15566842501","resolution":{"observed_at":"2026-08-12T04:51:16.999053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.003186Z","title":"GPT s Don ' t Keep Secrets: Searching for Backdoor Watermark Triggers in Autoregressive Language Models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.003186Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:680ffeef22663fbae1265509df67bb624b5cf400b19f507a18e52c4a2ded1aa1","observation_id":"977ce2f1-f298-4342-b774-2ceae3df2885","resolution":{"observed_at":"2026-08-12T04:51:17.003186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.007660Z","title":"Reliability Check: An Analysis of GPT -3's Response to Sensitive Topics and Prompt Wording","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.007660Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:af2e20b58164ae01640ff09982721480b3cc5ab9b0224c8c30c6a9e5fd000d91","observation_id":"855102a8-6674-4972-b483-eab017c5bbfb","resolution":{"observed_at":"2026-08-12T04:51:17.007660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.011852Z","title":"Driving Context into Text-to-Text Privatization","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.011852Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:3d9c9ccaa7f945e0531de26d9cd38fe6f00658e0a4ebaf02cc3603e2edaff1d2","observation_id":"a2e7835e-f913-4a22-b938-6112a387e6b0","resolution":{"observed_at":"2026-08-12T04:51:17.011852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.016058Z","title":"Expanding Scope: Adapting E nglish Adversarial Attacks to C hinese","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.016058Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:fdafcbb4817f655da119e9eb645d62a4a09221c69b26872314413425ca8dd4bd","observation_id":"a569e7ad-2d73-44ca-8d2e-99b7c9e94b1e","resolution":{"observed_at":"2026-08-12T04:51:17.016058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.020464Z","title":"Flatness-Aware Gradient Descent for Safe Conversational AI","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.020464Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:a8f19670044e41ef245f5ee5968b302c4cf8a4fcb2fb5bda464a366c64e83c58","observation_id":"0b3994e2-3d6e-410b-8a35-77b802cde3ba","resolution":{"observed_at":"2026-08-12T04:51:17.020464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.024785Z","title":"PBI -Attack: Prior-Guided Bimodal Interactive Black-Box Jailbreak Attack for Toxicity Maximization","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.024785Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:bb1d5620bc06cdf86ae382dc62848b5de7400a5c8434f9f8360b1645a152e949","observation_id":"8a824ce0-ecf5-4508-b1c1-5e2749353908","resolution":{"observed_at":"2026-08-12T04:51:17.024785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.028942Z","title":"Beyond Text-to- SQL for IoT Defense: A Comprehensive Framework for Querying and Classifying IoT Threats","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.028942Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:d991181eed82a1fa6421797de3a3922f3a1153d26d1bdb7cf740dfa719385b91","observation_id":"70156afd-43f2-4b56-9937-21f1a44382fc","resolution":{"observed_at":"2026-08-12T04:51:17.028942Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.033333Z","title":"Minimal Evidence Group Identification for Claim Verification","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.033333Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:5929327c6c2829c9b4c087c53f819e30e64edd2fe152068ca459723d54aea29e","observation_id":"43c6f142-7da7-4d57-95c6-a3c650c01f21","resolution":{"observed_at":"2026-08-12T04:51:17.033333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.037344Z","title":"Estimating Knowledge in Large Language Models Without Generating a Single Token","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.037344Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:8bd653e52229caa3c7e9cea7623b0732b3f7e91652d83f46940b6459793696bf","observation_id":"cd32d46d-f950-4093-ad1d-ba9016f142d7","resolution":{"observed_at":"2026-08-12T04:51:17.037344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.041415Z","title":"Intrinsic Test of Unlearning Using Parametric Knowledge Traces","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.041415Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:27002ed803ffa283133e9e3a33f83b4416a6ebda2b40e4dbdf8890439ed2d553","observation_id":"e0f4c745-e219-4e72-80c3-ed429bbe5883","resolution":{"observed_at":"2026-08-12T04:51:17.041415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.045477Z","title":"Can we trust the evaluation on C hat GPT ?","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.045477Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:3986d733b4338e5a1a057da7c84ad48fb4b54fbf2dc01f2d8e6cb5d0aeb7d211","observation_id":"fde013c0-e009-4fed-9000-b955d5bd1970","resolution":{"observed_at":"2026-08-12T04:51:17.045477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.049608Z","title":"Improving Factuality of Abstractive Summarization via Contrastive Reward Learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.049608Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:0101f2342e13d375959f823769c1adb04910017a11bb08956ae1a04f57a2cd6d","observation_id":"bb79b921-8c81-4537-bcb9-a161c973fd60","resolution":{"observed_at":"2026-08-12T04:51:17.049608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.053733Z","title":"Exploring Causal Mechanisms for Machine Text Detection Methods","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.053733Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:1e437ae3f1aaf0446318f7c8ab9ed787f5eb74698c1bca83df3ce29fc2737624","observation_id":"81f29426-d024-4297-bba4-f31f8c9d42e3","resolution":{"observed_at":"2026-08-12T04:51:17.053733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.057835Z","title":"On the Robustness of Agentic Function Calling","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.057835Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:790dc374a07a7892524c8d4d5aeee622a284a49395215326758d61a53e763247","observation_id":"29cfc788-99b7-49f6-9547-553936eebf07","resolution":{"observed_at":"2026-08-12T04:51:17.057835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.061990Z","title":"Cross-Task Defense: Instruction-Tuning LLM s for Content Safety","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.061990Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:2e0478fe433037f5dc2eaf5de9080c9372a2425f3d4013f9d0260c0fd6b03210","observation_id":"e27a5b00-a18a-48c3-8fc1-db57c08867f8","resolution":{"observed_at":"2026-08-12T04:51:17.061990Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.trustnlp-1.6","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.710054Z","title":"Gender Bias in Natural Language Processing Across Human Languages","venue":null,"work_id":"e943bf91-85e3-4575-bd3f-359f9864d6a5","year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.066494Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:792faae2436ee5ba364244b1d50894d1504bd69c034572c0d672220881aad2d1","observation_id":"d9e1eb20-7381-4116-8447-fdb5428c4c1e","resolution":{"observed_at":"2026-08-12T04:51:17.714511Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.071469Z","title":"Into the Gap between What Language Models Say and What They Know","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.071469Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:31a244ea69bedb434b4f87b159c7d6e544cfb8258992b2cdb742b53ffe9137dd","observation_id":"c4df4eb7-a254-42fa-a742-7af291b2e8df","resolution":{"observed_at":"2026-08-12T04:51:17.071469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.075909Z","title":"The False Sense of Privacy in LLM s: Non-Verbatim Memorization and Semantic Leakage","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.075909Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:a2cbe3533f12055e2fb441ddbbe1af630612db17be30f0ef5e27c35d19ce59db","observation_id":"1dd145f3-52cc-4863-b58e-f57ea78259a9","resolution":{"observed_at":"2026-08-12T04:51:17.075909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.081130Z","title":"and Raimundo, Marcos M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.081130Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:8e9450e2249d4505010a58a81c0bcf82b21da48ec391568bb302684d3ee00b69","observation_id":"665d41c0-4741-4d55-bfd6-a3f69d137367","resolution":{"observed_at":"2026-08-12T04:51:17.081130Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.085233Z","title":"2023 , howpublished =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":107,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.085233Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:498a8a20debf89af9ad01e4a88f0494735cca4f24cea2d6b379bd8fa8842f22d","observation_id":"df6bb622-cc75-4d30-a7c2-7f3f0cf08de8","resolution":{"observed_at":"2026-08-12T04:51:17.085233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.089472Z","title":"Nature Machine Intelligence , volume=","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":108,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.089472Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:994058ecc3490ebb6412b05bd8aaa8572418ee68052d3597255bd721dd0b3552","observation_id":"b8febde7-8661-42d9-a1ad-9d74840a86c0","resolution":{"observed_at":"2026-08-12T04:51:17.089472Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.113262Z","title":"Formalizing Trust in Artificial Intelligence: Prerequisites, Causes and Goals of Human Trust in AI , year =","venue":null,"work_id":"826ac402-e118-4c2b-bb0d-83b614d3d2e1","year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":109,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.093468Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:8afbee57bba7b69b667b4b5037ebb3eae1a64b8c44835a2a80759c531a7b885b","observation_id":"86f3659e-2cca-43b0-90de-d40d432325ad","resolution":{"observed_at":"2026-08-12T04:51:18.118140Z","resolver_source":"arxiv_id_nonexistent","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.097633Z","title":"FAccT 2022 , year =","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":110,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.097633Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:b3fd3293be2348216f3e5dcd9b4da24b4c1014ad3b098c531185ade1004aaf2f","observation_id":"4ae1e4f8-d7f1-4291-852c-520c7d802d47","resolution":{"observed_at":"2026-08-12T04:51:17.097633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.eacl-main.116","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.448217Z","title":"SODAPOP : Open-Ended Discovery of Social Biases in Social Commonsense Reasoning Models","venue":null,"work_id":"1feebea4-a47b-4751-b073-b36265b3bcf1","year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":111,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.102185Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:c8430b69cd17f5d0af829b3c1f4b065250da796cf012905dfeb3cd3a85928fad","observation_id":"0147a468-2552-4eda-a574-a7dee4c5c3ae","resolution":{"observed_at":"2026-08-12T04:51:17.453501Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.106463Z","title":"F air B elief - Assessing Harmful Beliefs in Language Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":112,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.106463Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:599c9bb891518540255d9ea60ba345cd6a03bb1645553c24dc7bd786e35547e6","observation_id":"a4a719c5-83ac-4805-bb0e-d9e500235ea1","resolution":{"observed_at":"2026-08-12T04:51:17.106463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.110769Z","title":"Investigating and Addressing Hallucinations of LLM s in Tasks Involving Negation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":113,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.110769Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:e65b7c30d844e62d297bdef7762b245a852d09020de3851a89fe68b346e59e9f","observation_id":"6f1b58e1-2ae7-4f1e-b79b-900131422633","resolution":{"observed_at":"2026-08-12T04:51:17.110769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.114972Z","title":"Introducing G en C eption for Multimodal LLM Benchmarking: You May Bypass Annotations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.114972Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:0561850611f71c54a666e5e8f87bce776e136e539393984d410666c6ce6742d3","observation_id":"b6c78b3f-8159-465f-af7a-ee3361fa9b89","resolution":{"observed_at":"2026-08-12T04:51:17.114972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.119168Z","title":"Tell Me Why: Explainable Public Health Fact-Checking with Large Language Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.119168Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:70402b22309061ec50205ef8423c6bc36d8a432075191e2b13b04f211c1efcda","observation_id":"1ae8427c-ff08-47d1-a8a4-5c26bd86b8ef","resolution":{"observed_at":"2026-08-12T04:51:17.119168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.123464Z","title":"Disentangling Linguistic Features with Dimension-Wise Analysis of Vector Embeddings","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":116,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.123464Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:8a0496cc76418a2a27d6b6394ceb692d34cf5beadc2a6850c2cd7a4fb67b7c9f","observation_id":"006a2779-9e8f-4f0a-bcfe-076eb44a3cef","resolution":{"observed_at":"2026-08-12T04:51:17.123464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.127721Z","title":"On The Real-world Performance of Machine Translation: Exploring Social Media Post-authors' Perspectives","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":117,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.127721Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:3e1a088678cdf7a68e2db21eece48738b09e6ddffec8ac0daa94d785c421c3fc","observation_id":"bb1c49fa-76ae-401e-b19e-adb654a3615f","resolution":{"observed_at":"2026-08-12T04:51:17.127721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.131935Z","title":"V i B e: A Text-to-Video Benchmark for Evaluating Hallucination in Large Multimodal Models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":118,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.131935Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:0423f2d22da52260c0041f703189b88e59103dc1ce36f0280652027e050fbaa3","observation_id":"3e020f1c-d738-4c26-8411-4ed746e324a8","resolution":{"observed_at":"2026-08-12T04:51:17.131935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.136528Z","title":"FACTOID : FAC tual en T ailment f O r halluc I nation Detection","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":119,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.136528Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:fbe684ba4f232a802d24b2e9f8de9afe2e6728ca2cfd9f47bcc86b136b9fbb6f","observation_id":"525ff90f-5571-4d9d-8c82-0760ff712954","resolution":{"observed_at":"2026-08-12T04:51:17.136528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.trustnlp-1.3","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.378228Z","title":"Private Release of Text Embedding Vectors","venue":null,"work_id":"c552e127-8fbd-437b-adff-8d5a52282755","year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":120,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.141076Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:6ddb99eccde5111285d5e568d734ffefe7ce835c9d47a6841680317f3c254212","observation_id":"90472fb8-a17a-4039-83a5-b2ab7fd06287","resolution":{"observed_at":"2026-08-12T04:51:17.384662Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.145503Z","title":"Challenges in Applying Explainability Methods to Improve the Fairness of NLP Models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.145503Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:83096a67bf5d3901c0f91d08f24b5d975327ceb05081213d7cce4cc8fb4ca4ae","observation_id":"74282c53-747a-475f-98ee-858b036da7a9","resolution":{"observed_at":"2026-08-12T04:51:17.145503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.149773Z","title":"An Encoder Attribution Analysis for Dense Passage Retriever in Open-Domain Question Answering","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":122,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.149773Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:387163bf559b7dbf0f27172d10127f41c1b1fd9f5941043aa767f93ca67c967f","observation_id":"2bb8e51e-980a-4099-8f0a-675e9c86307c","resolution":{"observed_at":"2026-08-12T04:51:17.149773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.154226Z","title":"A Keyword Based Approach to Understanding the Overpenalization of Marginalized Groups by E nglish Marginal Abuse Models on T witter","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":123,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.154226Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:8e14558fa23739c52ae38c5580ef7d5082a47e17c33aacfa5c55370407fe0077","observation_id":"b90ff77c-5bbe-4c31-85ed-7a757a7e202a","resolution":{"observed_at":"2026-08-12T04:51:17.154226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.537152Z","title":"Examining the Causal Impact of First Names on Language Models: The Case of Social Commonsense Reasoning","venue":null,"work_id":"715488fb-91a0-4811-a2db-a53e0bcc5596","year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":124,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.158335Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:31ab70b84a77555eb0047a232774babc8f7a924298b0da51bf2e2bea4b605cd2","observation_id":"adf32ab9-b7de-43be-b04c-19f465a8f73d","resolution":{"observed_at":"2026-08-12T04:51:18.541575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.523652Z","title":"An Empirical Study of Metrics to Measure Representational Harms in Pre-Trained Language Models","venue":null,"work_id":"025dda30-cec8-4ca9-99b7-3bebd9fd632a","year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.162510Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:76e2a02a9953b3ea79b41c5a0399981d42ea457a20f554e1a6ffa6faabc1b21a","observation_id":"9fa36f13-191e-494a-b98d-78ff65c182fa","resolution":{"observed_at":"2026-08-12T04:51:18.528229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.509924Z","title":"Beyond T uring: A Comparative Analysis of Approaches for Detecting Machine-Generated Text","venue":null,"work_id":"c512ca43-7fe2-41d4-9d80-aef2b338aa8e","year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":126,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.166956Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:cc13daa72ceedbe203ac523f156735f4d58e2091b2f89977d90a501a85935a6b","observation_id":"341e203f-54ec-4c36-8eb9-9c88008c453a","resolution":{"observed_at":"2026-08-12T04:51:18.514337Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.496178Z","title":"Automated Adversarial Discovery for Safety Classifiers","venue":null,"work_id":"9544340f-a1ce-4014-bf8b-5e5f8921e6ec","year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":127,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.171286Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:627f62562b73036fe403f5ee992f2030986a3412303e5969b256d4c89d57d1d8","observation_id":"005475b9-56ca-431d-848e-697fccbee092","resolution":{"observed_at":"2026-08-12T04:51:18.500532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.482427Z","title":"The Trade-off between Performance, Efficiency, and Fairness in Adapter Modules for Text Classification","venue":null,"work_id":"5a3afb37-1d8c-4a66-a182-054618c46ee5","year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":128,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.175925Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:9d0ce808e042dda1ec2f36018068e4821c83af3efd7b58d085882911a5a79c7a","observation_id":"7ea20418-09e4-482a-9562-f91ede53ab8f","resolution":{"observed_at":"2026-08-12T04:51:18.487078Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.468295Z","title":"On the Interplay between Fairness and Explainability","venue":null,"work_id":"1e539a36-3aca-4676-96f4-8f278309025c","year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.180480Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:6913ce99e020c0cd4c8d9de8e65afb20a78d898bdd735a1db51219f8f8e8b663","observation_id":"190fa7b7-a3c4-4d80-876e-5395f7ffa0f7","resolution":{"observed_at":"2026-08-12T04:51:18.473132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.453304Z","title":"F act A lign: Fact-Level Hallucination Detection and Classification Through Knowledge Graph Alignment","venue":null,"work_id":"fc9db4fc-6c46-4f7c-87d4-977873edf957","year":2024},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":130,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.185252Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:c2df5e035ca7c123fb6dfce769a71a941d32f55566e95ddf39b89bc86dcfae9c","observation_id":"55ba8d40-09b8-46fd-96da-4a6b551689e9","resolution":{"observed_at":"2026-08-12T04:51:18.458476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.438669Z","title":"Break the Breakout: Reinventing LM Defense Against Jailbreak Attacks with Self-Refine","venue":null,"work_id":"90b7d2d4-f836-4a2e-bace-e64bdab0992e","year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":131,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.190099Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:f73ac85b8ed63efcc44d562f79f8f28e0c86dce28167910fe8e24647c6a9a7e7","observation_id":"8c75aa3a-5794-4c63-88b4-404d2bc63ee8","resolution":{"observed_at":"2026-08-12T04:51:18.443147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.423902Z","title":"Ambiguity Detection and Uncertainty Calibration for Question Answering with Large Language Models","venue":null,"work_id":"4adfbc2f-c98b-49e8-bcdb-b210c48d8cab","year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":132,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.194565Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:dfc08a29e87d90781c859a792456af4f8df07a0645f1740c40db88abc3d85eb5","observation_id":"6f66e172-2292-46bb-b6a1-953db8cc11f4","resolution":{"observed_at":"2026-08-12T04:51:18.428494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.198931Z","title":"Error Detection for Multimodal Classification","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":133,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.198931Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:ec3d97e3ed50c74c110c79c6de280fb3a00c317b96c1a9bd48cd188740b2ae92","observation_id":"29146040-7016-4cff-8c6c-ea063beffc48","resolution":{"observed_at":"2026-08-12T04:51:17.198931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.410152Z","title":"Know What You do Not Know: Verbalized Uncertainty Estimation Robustness on Corrupted Images in Vision-Language Models","venue":null,"work_id":"7dec2fe1-bcee-4c87-aaae-3f0df2d498f1","year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":134,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.203373Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:cb795c28ac57c9f0077103e5c2dfa7463c4de5319478b565631316ab2dde02b5","observation_id":"91c8a12d-937d-4e08-ba69-3eec055fa2a8","resolution":{"observed_at":"2026-08-12T04:51:18.414831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.395492Z","title":"Multi-lingual Multi-turn Automated Red Teaming for LLM s","venue":null,"work_id":"f7e1651f-5f29-4b32-9c61-1ae81c1ded28","year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":135,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.207859Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:3bea8a70abef414a0dc92e2fbb3f950546444a1279b953c1a4df4463b8b45926","observation_id":"a7212c6c-36cd-4ffb-b00b-7e7f9fc293b7","resolution":{"observed_at":"2026-08-12T04:51:18.400454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:18.381243Z","title":"Line of Duty: Evaluating LLM Self-Knowledge via Consistency in Feasibility Boundaries","venue":null,"work_id":"f521a5c5-6fcd-4f07-8dbc-287a86413caf","year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":136,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.212116Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:1fb9de5a1a10c77d6bf332c9d5ba719e6c6ed1a72e47a65d2480cf2bcbcf060e","observation_id":"5be57346-eaa7-4f20-8844-ef8f0db1a5bf","resolution":{"observed_at":"2026-08-12T04:51:18.385972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11133","last_updated":"2025-09-03T17:03:15Z","snapshot_observed_at":"2026-08-19T05:51:00.536266Z","submitted_at":"2025-08-15T00:58:10Z","title":"MoNaCo: More Natural and Complex Questions for Reasoning Across Dozens of Documents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.11133","snapshot_observed_at":"2026-08-12T04:51:17.216216Z","title":"M o N a C o: More Natural and Complex Questions for Reasoning Across Dozens of Documents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":137,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.216216Z"},"links":{"cited_paper":"/paper/2508.11133","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:3c5a24c0f588b3884b6f58065f54bffcb39256e3e6722b34751ff0167cf06812","observation_id":"4fb8417f-d540-425e-9264-e7743f4588b2","resolution":{"observed_at":"2026-08-12T04:51:17.216216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.220598Z","title":"and Aletras, Nikolaos and Ma, Ning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":138,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.220598Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:c601c06aad4f8e52cc61578d1ffeec5491d98c1ef9418f1bbd655461de61e86b","observation_id":"54f6a97a-3cd1-4e93-afd0-4e4e809588cf","resolution":{"observed_at":"2026-08-12T04:51:17.220598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.14168","last_updated":"2021-12-28T14:54:18Z","snapshot_observed_at":"2026-08-19T17:56:34.986651Z","submitted_at":"2021-12-28T14:54:18Z","title":"A Survey on Gender Bias in Natural Language Processing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.14168","snapshot_observed_at":"2026-08-12T04:51:17.224833Z","title":"A Survey on Gender Bias in Natural Language Processing","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":139,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.224833Z"},"links":{"cited_paper":"/paper/2112.14168","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:927f3f65f6a69aa57154fa69ee92f9422f09a8d9a260cdff4b36a2282bd31180","observation_id":"7046bf1a-afc2-43c3-b9cf-0cafdae14f5b","resolution":{"observed_at":"2026-08-12T04:51:17.224833Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.229554Z","title":"Inducing Positive Perspectives with Text Reframing","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":140,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.229554Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:be8fd58ad6b7efd81fd47ab827e51cb3af3cbf6035527a0741e1d2b99f3ab16f","observation_id":"443dba07-8f5f-40a6-8d92-ae7ad2edc7fa","resolution":{"observed_at":"2026-08-12T04:51:17.229554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.233451Z","title":"The Importance of Modeling Social Factors of Language: Theory and Practice","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":141,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.233451Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:d91e799cc1ce30ee880e2a4289dca1926dee374b8fd6b90dc32eadffcbedd0a1","observation_id":"a71ace5c-74d1-4d18-9a9d-2fe24ea8f402","resolution":{"observed_at":"2026-08-12T04:51:17.233451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-12T04:51:17.237236Z","title":"arXiv preprint arXiv:2307.09288 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":142,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.237236Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:0f4ddb9176adf500fda3cd151a4da982cb6e8bfc52fd1959bdb70d9886a01e95","observation_id":"1e9ce44b-5ca1-4dc6-b711-3998d2ef9c6d","resolution":{"observed_at":"2026-08-12T04:51:17.237236Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.241762Z","title":"2023 , howpublished =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":143,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.241762Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:5d35ad080306953c4a87a5f2b6a6a9e65808bd45906abac006ea0072894645e6","observation_id":"258c868f-edfa-44fd-8a50-a06277381181","resolution":{"observed_at":"2026-08-12T04:51:17.241762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T04:51:17.246222Z","title":"arXiv preprint arXiv:2303.08774 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":144,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.246222Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:8be8244b1985e6db9fcd1848e9a5d9165d43c4c7c659d01b3e0a0d7126b76345","observation_id":"7650e705-1670-4fde-aa0c-ef1a3c49304a","resolution":{"observed_at":"2026-08-12T04:51:17.246222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.250929Z","title":"Strength in Numbers: Estimating Confidence of Large Language Models by Prompt Agreement","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":145,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.250929Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:6fcb003189023410d6f235bed9aecbda0817da563abd75bada6649bd1ef267ca","observation_id":"4a4c4ba2-69bc-47c8-92bb-119dbf2d7bf4","resolution":{"observed_at":"2026-08-12T04:51:17.250929Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.255122Z","title":"On the Intrinsic and Extrinsic Fairness Evaluation Metrics for Contextualized Language Representations","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":146,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.255122Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:828cc40fdacc16f00cf66703dcdeebf106540be1deb80d595f055fd090f44aea","observation_id":"4617b8f3-ce41-4840-a759-dd30de6748b9","resolution":{"observed_at":"2026-08-12T04:51:17.255122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T04:51:17.259117Z","title":"Pay Attention to the Robustness of C hinese Minority Language Models! Syllable-level Textual Adversarial Attack on T ibetan Script","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop","version":1},"reference_index":147,"source":"arxiv_source","source_observed_at":"2026-08-12T04:51:17.259117Z"},"links":{"citing_paper":"/paper/2608.11171"},"observation_digest":"sha256:62aa1aa9621dddf4eb265f4d1043baa507e3ae02fd472e29b1bc227bb4e3ebf5","observation_id":"9b6dd65d-2dd3-4dc9-b107-a6e7c1e3b72c","resolution":{"observed_at":"2026-08-12T04:51:17.259117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.11171","last_updated":"2026-08-11T17:30:16Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-17T22:58:51.112378Z","submitted_at":"2026-08-11T17:30:16Z","title":"From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":82,"verified_exact":6,"verified_fuzzy":12},"total_outbound_references":118},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 100 of 118 outbound references and 0 inbound Pith citation observations for arXiv:2608.11171."}