{"as_of":"2026-08-23T12:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7ec6c205bcfd3641e1505bdd38cef0a96833b71a420be2cffee83081daf7493b","coverage":[{"denominator":77,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":77,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T21:11:02.149478Z","state":"measured"},{"denominator":80,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":80,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:51:56.450815Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-17T00:31:24.634056Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.10670","snapshot_observed_at":"2026-08-07T04:51:56.450815Z","title":"Interpretable risk mitigation in llm agent systems.arXiv preprint arXiv:2505.10670, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.09420","last_updated":"2025-06-11T06:08:13Z","snapshot_observed_at":"2026-08-09T00:55:40.500387Z","submitted_at":"2025-06-11T06:08:13Z","title":"A Call for Collaborative Intelligence: Why Human-Agent Systems Should Precede AI Autonomy","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T04:51:56.450815Z"},"links":{"cited_paper":"/paper/2505.10670","citing_paper":"/paper/2506.09420"},"observation_digest":"sha256:ce9e691a59f0c765e1e5e4b4af0c7f11a4f324957fd9078aa4e801d947ebea27","observation_id":"b72a9458-11dc-47d1-84f0-f77d9122b3d6","resolution":{"observed_at":"2026-08-07T04:51:56.450815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"cited_work":{"arxiv_id":"2505.10670","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.10670","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","venue":null,"work_id":"d2817880-ea6c-45a7-9d7d-8469adabd169","year":2025},"citing_paper":{"arxiv_id":"2512.05929","last_updated":"2026-06-29T18:18:24Z","snapshot_observed_at":"2026-08-16T07:11:14.391624Z","submitted_at":"2025-12-05T18:12:21Z","title":"LLM Harms: A Taxonomy and Discussion","version":2},"reference_index":189,"source":"pdf_text","source_observed_at":"2026-05-17T00:29:07.951709Z"},"links":{"cited_paper":"/paper/2505.10670","citing_paper":"/paper/2512.05929"},"observation_digest":"sha256:eb6cc424930bd73eca14e3b202775dd85faa85436123246c906b31a4a92ddc77","observation_id":"1b7081ac-e99d-43fc-be2e-190be39eed65","resolution":{"observed_at":"2026-05-17T00:31:24.636618Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.10670","snapshot_observed_at":"2026-08-03T18:19:27.337242Z","title":"Interpretable Risk Mitigation in LLM Agent Systems,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.05929","last_updated":"2026-06-29T18:18:24Z","snapshot_observed_at":"2026-08-16T07:11:14.391624Z","submitted_at":"2025-12-05T18:12:21Z","title":"LLM Harms: A Taxonomy and Discussion","version":4},"reference_index":189,"source":"pdf_text","source_observed_at":"2026-08-03T18:19:27.337242Z"},"links":{"cited_paper":"/paper/2505.10670","citing_paper":"/paper/2512.05929"},"observation_digest":"sha256:4e62376da4334efae82eb7aaccbb4096187123306b539a8647ffc59c3a41853b","observation_id":"84d058c8-51a1-4859-970e-8b0c97117aac","resolution":{"observed_at":"2026-08-03T18:19:27.337242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.10670/citation-record","integrity":"/paper/2505.10670/integrity","json":"/paper/2505.10670/citation-record.json","paper":"/paper/2505.10670"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.638268Z","title":"Artificial intelligence and the future of work: Evidence from OECD countries","venue":null,"work_id":"4e93b0ea-6d02-427d-82c6-c8da5df0ea01","year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.782535Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:79993e8c21d8c5e9637f2f54e4a51bec726eb27fc4d15f3a28f316814ed9d414","observation_id":"c43eafac-f59d-4479-8e8e-f89c0279a822","resolution":{"observed_at":"2026-08-15T21:11:03.643143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.622368Z","title":"Dai, Chelsea Finn, Justin Fu, Kanishka Gopalakrishnan, et al","venue":null,"work_id":"4cfae7f4-561d-41bc-a0d9-7f211940f04a","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.787366Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:1fad4d05c22d3f8a2137a5c4aaf35c65ce807a07d8b10ad187002ad550da5d67","observation_id":"bf3d3a32-c723-4ab7-a21a-243b73e9489b","resolution":{"observed_at":"2026-08-15T21:11:03.627827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.607866Z","title":"Mistral 7b: Open foundation models, 2023","venue":null,"work_id":"d52424b5-fcf8-4fd8-900b-9ce56746b04f","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.792100Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:7e61a1c2686a9163be4f944f831ef10a2eaf1c0a4001555deeeb1e8f04360060","observation_id":"3801ed77-8037-4a96-b87d-1b592e49f25f","resolution":{"observed_at":"2026-08-15T21:11:03.612474Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16867","last_updated":"2025-05-07T12:44:45Z","snapshot_observed_at":"2026-08-18T11:09:48.072216Z","submitted_at":"2023-05-26T12:17:59Z","title":"Playing repeated games with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16867","snapshot_observed_at":"2026-08-15T21:11:01.796811Z","title":"Playing repeated games with large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.796811Z"},"links":{"cited_paper":"/paper/2305.16867","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:d3afb41abea193e6c233643ca65ccf524f447a9bd6e8ee50dead64e1a143b91e","observation_id":"54ed5ba8-5bf9-4e48-a28b-7b79fb7685e8","resolution":{"observed_at":"2026-08-15T21:11:01.796811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1606.06565","last_updated":"2016-07-25T17:23:29Z","snapshot_observed_at":"2026-07-06T05:00:46.434335Z","submitted_at":"2016-06-21T13:37:05Z","title":"Concrete Problems in AI Safety","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1606.06565","snapshot_observed_at":"2026-08-15T21:11:01.802560Z","title":"Concrete problems in AI safety","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.802560Z"},"links":{"cited_paper":"/paper/1606.06565","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e3fa78928aa155f3c7d2ed2d961082ed1d210699d78eb27984746a8e23c33479","observation_id":"9b7eff9b-d167-419d-ae41-da6ca8c46fe7","resolution":{"observed_at":"2026-08-15T21:11:01.802560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.592741Z","title":null,"venue":null,"work_id":"794a2695-9d64-47cf-9316-2d7aeafb5ec7","year":1984},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.807791Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:de24e546708778568d5d54e8027044342ab5aa5b64ae8c5a6d5562fc7db4e6a3","observation_id":"16702f8c-2cb7-4d44-a2c2-8387f4b0ae83","resolution":{"observed_at":"2026-08-15T21:11:03.597680Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-15T21:11:01.813065Z","title":"Training a helpful and harmless assistant with reinforcement learning from human feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.813065Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:865726595d338e7faf825c08db1da58d2aa98603e098f091e62f8b5a15e282ae","observation_id":"94b9346b-084f-4af5-a678-a12f9bd12757","resolution":{"observed_at":"2026-08-15T21:11:01.813065Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.575085Z","title":"Emergent tool use from multi-agent autocurricula","venue":null,"work_id":"dd733df4-15f3-4a2b-9ae8-50f4e56b74e0","year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.817697Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:a0d23ec8518990ddcca79e2efafcd3507d143a51f492cd75177acdaa95f3993e","observation_id":"5e6eafe6-a930-4240-a93e-aaa30f5aa5d6","resolution":{"observed_at":"2026-08-15T21:11:03.580794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.556424Z","title":"Bender, Timnit Gebru, Angelina McMillan-Major, and Shmargaret Shmitchell","venue":null,"work_id":"7eebd975-9ca2-4abc-8f1e-d2e40b0611f0","year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.822090Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:467e556e89330b1c76165510611195219cf498d88fd3e3b94c8ec67492542526","observation_id":"351ad7f1-6ab1-403d-b7ff-ef072b69c117","resolution":{"observed_at":"2026-08-15T21:11:03.562891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07258","last_updated":"2022-07-12T23:45:14Z","snapshot_observed_at":"2026-08-02T09:20:40.804790Z","submitted_at":"2021-08-16T17:50:08Z","title":"On the Opportunities and Risks of Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07258","snapshot_observed_at":"2026-08-15T21:11:01.826932Z","title":"Hudson, Ehsan Adeli, Russ Altman, Simran Arora, Sydney von Arx, et al","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.826932Z"},"links":{"cited_paper":"/paper/2108.07258","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:8578c533134ec3310aaabd602d5652d7eeeac1d59da5094c9fd9276d01fbc42d","observation_id":"8a106779-c66d-48c7-8c48-b0f78d8abf43","resolution":{"observed_at":"2026-08-15T21:11:01.826932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.538368Z","title":null,"venue":null,"work_id":"cdf4c402-e4e3-47b5-92a9-7832bfd81d54","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.832003Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:89b4b36f505750537dc380559a5c73360d9e2e8dee99323c2e83b0b3040cce3b","observation_id":"2a7adc57-b623-480d-be0c-b1dd7e69a6ee","resolution":{"observed_at":"2026-08-15T21:11:03.544642Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.06817","last_updated":"2023-08-11T17:45:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-12-13T18:55:15Z","title":"RT-1: Robotics Transformer for Real-World Control at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.06817","snapshot_observed_at":"2026-08-15T21:11:01.836214Z","title":"RT-1: Robotics transformer for real-world control at scale","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.836214Z"},"links":{"cited_paper":"/paper/2212.06817","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:7271eb10f1c4ecad6f70d0e33ba6e6d2053f72c397f9f2aa5f09cc6785ffa3e9","observation_id":"a338b78c-533d-4e4b-9b30-639f80356236","resolution":{"observed_at":"2026-08-15T21:11:01.836214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.521438Z","title":"Playing games with gpt: What can we learn about a large language model from canonical strategic games? SSRN Electronic Journal, 2023","venue":null,"work_id":"daec497b-2379-4a00-acba-e406c24b6c01","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.840725Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e1f1c021b901035a7cc332740732fe974a9d57a98a9fa7b85f3a8168a58e8133","observation_id":"16d22cc2-e43f-4ec2-aff9-61a6d878a1a6","resolution":{"observed_at":"2026-08-15T21:11:03.526152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.503212Z","title":"Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D","venue":null,"work_id":"2ddb56ad-1b11-4050-a6e4-1cf601cd82dd","year":1901},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.844793Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:3a4d8943b999799733c9c518607e3adcf86817e56259f6daed6d86deca60439a","observation_id":"fa8b734c-b330-4c95-85f8-426bab50f0df","resolution":{"observed_at":"2026-08-15T21:11:03.508895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.485636Z","title":"What can machine learning do? workforce implications","venue":null,"work_id":"e4531483-9fe0-4eb4-9335-4f27f6b3199e","year":2017},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.848928Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:69b331cdb61985dcdebed6f2af1dcaa627ff50957d6142eeebeaa4a000f82394","observation_id":"b60e728a-df0f-4060-9aeb-3d455d436d74","resolution":{"observed_at":"2026-08-15T21:11:03.491497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-20T20:50:23.483838Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-15T21:11:01.853101Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.853101Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:44563a09fdfb88c9d8e1e5ecc642bad3748a17bf2f4a7e96d8abfdf628990897","observation_id":"e55a86e7-747f-4031-8209-67c0bab512ab","resolution":{"observed_at":"2026-08-15T21:11:01.853101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.857296Z","title":"Instigating cooperation among llm agents using adaptive information modulation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.857296Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:2c172eae16cfd41558585d7147c0f6115041eef4f8fe146ad3c44c2c89e8b79f","observation_id":"08b4f22d-3581-40e0-aa27-dd14aaa9041a","resolution":{"observed_at":"2026-08-15T21:11:01.857296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.08600","last_updated":"2023-10-04T13:17:38Z","snapshot_observed_at":"2026-08-21T17:39:12.921162Z","submitted_at":"2023-09-15T17:56:55Z","title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.08600","snapshot_observed_at":"2026-08-15T21:11:01.870732Z","title":"Sparse autoencoders find highly interpretable features in language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.870732Z"},"links":{"cited_paper":"/paper/2309.08600","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6bba13bfaf3ef1ffd34712d97a897e626cac3729b45f2ce8a024d1b962aea5ca","observation_id":"72433d46-a595-481b-a5ef-2f80a6405e69","resolution":{"observed_at":"2026-08-15T21:11:01.870732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.453516Z","title":"Reinforcement learning in a prisoner’s dilemma","venue":null,"work_id":"c52095a1-54e2-45ff-97a0-fa0936d831fd","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.875550Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e0d3699950efb846091f5fea687d5f5e8629ca70e757ee66d929e39c48940e4f","observation_id":"eadd38c1-fbcd-4b04-a105-2ea6f1e6a5e6","resolution":{"observed_at":"2026-08-15T21:11:03.458432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.10652","last_updated":"2022-09-21T20:49:26Z","snapshot_observed_at":"2026-08-22T06:24:21.120481Z","submitted_at":"2022-09-21T20:49:26Z","title":"Toy Models of Superposition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.10652","snapshot_observed_at":"2026-08-15T21:11:01.880133Z","title":"Toy models of superposition","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.880133Z"},"links":{"cited_paper":"/paper/2209.10652","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:06624c60b4b516fca0715b1d251458e8c8b6a5b9eff407308462355035a7ff7a","observation_id":"24f6a202-78a8-481d-8db8-1c4b79df229d","resolution":{"observed_at":"2026-08-15T21:11:01.880133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.04866","last_updated":"2022-12-19T17:54:22Z","snapshot_observed_at":"2026-08-23T01:56:58.965972Z","submitted_at":"2022-10-10T17:34:49Z","title":"PoGaIN: Poisson-Gaussian Image Noise Modeling from Paired Samples","version":2},"cited_work":{"arxiv_id":"2210.04866","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.04866","snapshot_observed_at":"2026-08-15T21:11:02.828852Z","title":"PoGaIN: Poisson-Gaussian Image Noise Modeling from Paired Samples","venue":"cs.CV","work_id":"fa2a04ea-1db8-4f70-933c-c9bce17001f1","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.886176Z"},"links":{"cited_paper":"/paper/2210.04866","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:b608c7d9f214673e01d21f596742d744fafd42916ae0999a25485aac42091eb4","observation_id":"3da5c266-6d9b-45d6-8d90-16eafac12480","resolution":{"observed_at":"2026-08-15T21:11:02.834041Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14860","last_updated":"2025-02-27T03:03:59Z","snapshot_observed_at":"2026-08-19T12:35:11.476497Z","submitted_at":"2024-05-23T17:59:04Z","title":"Not All Language Model Features Are One-Dimensionally Linear","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14860","snapshot_observed_at":"2026-08-15T21:11:01.891128Z","title":"Michaud, Wes Gurnee, and Max Tegmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.891128Z"},"links":{"cited_paper":"/paper/2405.14860","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:2d47d6f0b37d363b98c45979e3897450068c4b4325ddb2ca80cef4c80cb38a09","observation_id":"9a93bcfe-f4cc-4a52-9e56-a716c5208000","resolution":{"observed_at":"2026-08-15T21:11:01.891128Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.438321Z","title":"Some experimental games","venue":null,"work_id":"25e408a7-0f48-4cb6-bfaa-49df36c93acb","year":1958},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.896121Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:0e185a2782d5535a4d35a58229ad3c64118b6c039e2d09489bae5994e6aff818","observation_id":"80ab46e3-96c8-4341-b15b-4ef484f62fc1","resolution":{"observed_at":"2026-08-15T21:11:03.442507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13605","last_updated":"2024-09-19T15:19:58Z","snapshot_observed_at":"2026-08-21T05:39:01.272823Z","submitted_at":"2024-06-19T14:51:14Z","title":"Nicer Than Humans: How do Large Language Models Behave in the Prisoner's Dilemma?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13605","snapshot_observed_at":"2026-08-15T21:11:01.900891Z","title":"Nicer than humans: How do large language models behave in the prisoner’s dilemma? ArXiv, abs/2406.13605, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.900891Z"},"links":{"cited_paper":"/paper/2406.13605","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e3876cc9b7a6b030a24c1693253c6dee8d6736f75e6f14146152227bc6798e88","observation_id":"f9da16f4-a983-41b4-b6d2-cccdb6ba0154","resolution":{"observed_at":"2026-08-15T21:11:01.900891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.906112Z","title":"Artificial intelligence, values, and alignment","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.906112Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:2ec6d821156e684988dfa9802ea6f061515361cc506b385f2c54a760caf4cf62","observation_id":"88637004-c8d5-4677-890d-d755e3357d38","resolution":{"observed_at":"2026-08-15T21:11:01.906112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-15T21:11:01.910539Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.910539Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:b1c06bb7abfceeb4f9cd31334268027730481af1911091c6295ed83d019e970d","observation_id":"cb688a1f-ae7d-4316-9fcd-3ddffe2c900b","resolution":{"observed_at":"2026-08-15T21:11:01.910539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-15T21:11:01.915548Z","title":"Measuring massive multitask language understanding, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.915548Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:db44e3c4c4758bd5ff4548d4106a78bde799533a9d41cddaf82a82fe846a08c3","observation_id":"2aea34d7-c236-4ba2-b05c-5266d05cbe3a","resolution":{"observed_at":"2026-08-15T21:11:01.915548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.920013Z","title":"Measuring mathematical problem solving with the math dataset,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.920013Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:d23741fe8de883dfd9d3b1ce648611561b1e72942d6e0d55247ec967fff16180","observation_id":"daa69867-514e-4500-ac68-9b455cec0cf3","resolution":{"observed_at":"2026-08-15T21:11:01.920013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.411699Z","title":"Cogagent: A visual language model for gui agents","venue":null,"work_id":"ef5b7c60-0d6e-49bb-867b-689d2084ff61","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.930482Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:5d9eb1912aa5c15811c542c7ed0741d804423ac082d3c90b4c209dce26e0be4b","observation_id":"ccc257f3-639b-4bd7-b693-d99e2d88ec53","resolution":{"observed_at":"2026-08-15T21:11:03.416984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.395348Z","title":"Non-linear inference time intervention: Improving llm truthfulness","venue":null,"work_id":"64b1abf9-c196-413a-85b6-c036384036ef","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.935781Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:44d7949477bbc5c3f1954b24d84e2bf2712b9fc76c984b422e44e844dcc04e39","observation_id":"f465e08f-8581-42e6-b51b-25e319bc463c","resolution":{"observed_at":"2026-08-15T21:11:03.400826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10403","last_updated":"2023-05-26T17:59:33Z","snapshot_observed_at":"2026-08-06T19:23:30.079167Z","submitted_at":"2022-12-20T16:29:03Z","title":"Towards Reasoning in Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10403","snapshot_observed_at":"2026-08-15T21:11:01.939936Z","title":"Towards reasoning in large language models: A survey, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.939936Z"},"links":{"cited_paper":"/paper/2212.10403","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:cb56251b99a829a0cee8e561ff681bf805f4c508094740407e520e1066d319fe","observation_id":"79c959d9-a3b3-4327-86ed-96acb55e005b","resolution":{"observed_at":"2026-08-15T21:11:01.939936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.379909Z","title":"Large language models for uavs: Current state and pathways to the future","venue":null,"work_id":"7c2885cd-da7b-49bf-b8b4-ea6021e14e35","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.944592Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:86aac816c8485736c754ff53320a8f88ab8c941ef948b2fcd694e1d923669143","observation_id":"f8bf0ff6-e450-4f1c-b07c-950516506ed9","resolution":{"observed_at":"2026-08-15T21:11:03.384589Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-23T11:47:44.729242Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-15T21:11:01.949339Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.949339Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:4e7252137127b84c21b9257f59bedaa5f57fa74a92bd8f78d71da4ea5e65755c","observation_id":"a7a39f6f-26fb-4c28-b196-bc2a237200ff","resolution":{"observed_at":"2026-08-15T21:11:01.949339Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.360951Z","title":"llama-3-8b-it-res (revision 53425c3), 2024","venue":null,"work_id":"3a878928-d5e4-46ed-a09f-8b9fbf0b4061","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.953994Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:f1785714d4443c972c51c5e671505dab55ee397f9acd92e7664eee21eb06e2e6","observation_id":"017ef2a9-5520-4eb9-bdc4-f92641554c31","resolution":{"observed_at":"2026-08-15T21:11:03.367736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.345434Z","title":null,"venue":null,"work_id":"1cd1b422-42e2-4551-95eb-1494d96c94b5","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.958191Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:675e21684b51dc9ad019f2afcb429a4211bf3fe50bd39ffd165d96ecc2e2d1a7","observation_id":"91339d58-f620-44d4-8eb7-b59703b0d07a","resolution":{"observed_at":"2026-08-15T21:11:03.350181Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.328985Z","title":"Martin, Hans-Theo Normann, and T","venue":null,"work_id":"880a825f-4dc6-4839-8693-140a2986625c","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.962360Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:226f31da3436f6a9408bb2b42b0080a23bc26012dfb41eeb005bb806d9cd5b37","observation_id":"1a302ee4-fd9f-46ae-8fab-c55bc4736872","resolution":{"observed_at":"2026-08-15T21:11:03.333745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.313485Z","title":"Rusu, Kieran Milan, John Quan, Tiago Ramalho, Agnieszka Grabska-Barwinska, et al","venue":null,"work_id":"d514e000-10d5-4f79-b13b-835ac92a4b4b","year":2017},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.966629Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:185006412421ca7e7321b2e924dab76219679a07bc0658dc951323a761340743","observation_id":"71a8d99b-1365-4a5b-ac7b-cce9c7fceaf8","resolution":{"observed_at":"2026-08-15T21:11:03.318008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03341","last_updated":"2024-06-26T14:11:53Z","snapshot_observed_at":"2026-08-13T19:18:46.730073Z","submitted_at":"2023-06-06T01:26:53Z","title":"Inference-Time Intervention: Eliciting Truthful Answers from a Language Model","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03341","snapshot_observed_at":"2026-08-15T21:11:01.971070Z","title":"Inference-time intervention: Eliciting truthful answers from a language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.971070Z"},"links":{"cited_paper":"/paper/2306.03341","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:5fff60f8abf3393e8a84a90f1c0b36b9c287e7558800700d3ad7eb78b3b0e19b","observation_id":"53ae654e-f604-43fa-bec1-7e4bbe48d9bf","resolution":{"observed_at":"2026-08-15T21:11:01.971070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05147","last_updated":"2024-08-19T07:51:05Z","snapshot_observed_at":"2026-08-15T15:50:24.074149Z","submitted_at":"2024-08-09T16:06:42Z","title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05147","snapshot_observed_at":"2026-08-15T21:11:01.975880Z","title":"Gemma scope: Open sparse autoencoders everywhere all at once on gemma 2, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.975880Z"},"links":{"cited_paper":"/paper/2408.05147","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e26d8a59d2abe7d9479e73939754b62caf36db4ac1b45dbf2f5b97e66c155f2d","observation_id":"accf6756-7ccc-46fa-8ac0-5a95cac9c7e1","resolution":{"observed_at":"2026-08-15T21:11:01.975880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.297754Z","title":"The mythos of model interpretability","venue":null,"work_id":"1da3f0f0-5edb-4d67-8d1f-43d5f4c3e2b8","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.980577Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6aed82cb480083803288e57ee17c0003f7efe44c6d16dc260e1fdb2b1a4b71eb","observation_id":"e1e40780-c265-4c8d-b588-0f4db9b277d5","resolution":{"observed_at":"2026-08-15T21:11:03.302405Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.989753Z","title":"Pre-train, prompt, and predict: A systematic survey of prompting methods in natural language processing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.989753Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:64b120144d89a692168b6618e726cd86bd378cdbd72c11ffe167b23f67b2ebf0","observation_id":"4fc9c058-9b7f-4030-b591-6b05a8f26e5f","resolution":{"observed_at":"2026-08-15T21:11:01.989753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.05241","last_updated":"2024-10-30T18:37:57Z","snapshot_observed_at":"2026-08-16T13:28:20.917188Z","submitted_at":"2024-08-05T20:49:48Z","title":"Large Model Strategic Thinking, Small Model Efficiency: Transferring Theory of Mind in Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.05241","snapshot_observed_at":"2026-08-15T21:11:01.994175Z","title":"Large model strategic thinking, small model efficiency: Transferring theory of mind in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.994175Z"},"links":{"cited_paper":"/paper/2408.05241","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:0e243c61f1e3e23cdafff7cfc3578c4d9d3d68fb76472af60c0b47255017c2d0","observation_id":"3cdffd23-9790-4415-81f6-49a0963613c8","resolution":{"observed_at":"2026-08-15T21:11:01.994175Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.254557Z","title":"Linguistic regularities in continuous space word representations","venue":null,"work_id":"b53c9197-1afd-4023-96d6-b5b91c54f5d5","year":2013},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.003164Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:cf8f9ca6920797cd92290b7dc1e1ad40934ec4bb447f2c47d67e74eec9f4bd8a","observation_id":"c86296d2-ed87-49d3-b9f2-78d6f82f7afa","resolution":{"observed_at":"2026-08-15T21:11:03.259770Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06196","last_updated":"2025-03-23T14:51:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-09T05:37:09Z","title":"Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06196","snapshot_observed_at":"2026-08-15T21:11:02.008126Z","title":"Large language models: A survey, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.008126Z"},"links":{"cited_paper":"/paper/2402.06196","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:37d7137a026c8eca46c3a36ba52fcb1abb0d7d8ea860dc0150a9b80ad4c0d27b","observation_id":"a59a7951-c2e9-43c9-a557-199380bc4da6","resolution":{"observed_at":"2026-08-15T21:11:02.008126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.013053Z","title":"A strategy of win-stay, lose-shift that outperforms tit-for-tat in the prisoner’s dilemma game","venue":null,"work_id":null,"year":1993},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.013053Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:91d3b15568bc02d4be803b8c837e41a4b7fa497c5688fdada2e1522d4e021a27","observation_id":"fae7cda6-218c-4394-8746-aa9f8ca079ab","resolution":{"observed_at":"2026-08-15T21:11:02.013053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-15T21:11:02.018166Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, et al","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.018166Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:b63bfa43074214c2b69cd51995d26e965c1e47781566b87bd0f3e1c9077b07ab","observation_id":"4826b429-d15b-417a-b168-ff52c95b5cbc","resolution":{"observed_at":"2026-08-15T21:11:02.018166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"shsconf/2023178","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.602490Z","title":"Cooperation: A systematic review of how to enable agent to circumvent the prisoner’s dilemma","venue":null,"work_id":"4c42dc09-38e2-43a4-8bfd-3dd064e67328","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.023478Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6cf256f5a317fcba75e7c70e18adb5205949aa64cbcd47b101d1547201270b83","observation_id":"014ab367-984a-479d-83c0-103ef33837a1","resolution":{"observed_at":"2026-08-15T21:11:02.613607Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03442","last_updated":"2023-08-06T00:21:19Z","snapshot_observed_at":"2026-08-16T03:49:04.888755Z","submitted_at":"2023-04-07T01:55:19Z","title":"Generative Agents: Interactive Simulacra of Human Behavior","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03442","snapshot_observed_at":"2026-08-15T21:11:02.028670Z","title":"Wang, Linxi Wang, Alex Wang, Allie He, Qian Liao, David Kempe, et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.028670Z"},"links":{"cited_paper":"/paper/2304.03442","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:906119f1bbadc6f5573cf1b79034f51a28b0795aa8a0d02e5095cce36e62b730","observation_id":"7a05f4db-7afc-4148-a98b-9ef26cb1000d","resolution":{"observed_at":"2026-08-15T21:11:02.028670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11871","last_updated":"2025-05-21T06:52:31Z","snapshot_observed_at":"2026-08-20T06:33:09.979073Z","submitted_at":"2024-10-09T12:06:43Z","title":"TinyClick: Single-Turn Agent for Empowering GUI Automation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11871","snapshot_observed_at":"2026-08-15T21:11:02.033414Z","title":"Tinyclick: Single-turn agent for empowering gui automation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.033414Z"},"links":{"cited_paper":"/paper/2410.11871","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:16b75ff475d0a997bfab46a863d2da08f3618c86dd3bc96e88f511a4bd1a0acf","observation_id":"de77e093-1471-462a-b0a6-a927f6729dc5","resolution":{"observed_at":"2026-08-15T21:11:02.033414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.226853Z","title":null,"venue":null,"work_id":"4b071fc5-ec94-4935-ba55-1533214252dd","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.038063Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:741889572732ffe27e30b98c9071db38b31a482b300551c77d7d40e2e69354ff","observation_id":"73a4ef3f-38c0-44df-85f6-445319704ac3","resolution":{"observed_at":"2026-08-15T21:11:03.231251Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.211816Z","title":"Effect of private deliberation: Deception of large language models in game play","venue":null,"work_id":"b2f8b616-0fa5-494f-b414-6d88986d72fd","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.042680Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:044aaaae570b590ad668e41ac9b4559c57e8dbb3a33b1ca946854a425d2395bd","observation_id":"35196dd2-37eb-4576-a6c2-b40e2da03ea4","resolution":{"observed_at":"2026-08-15T21:11:03.216654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-21T17:04:32.090035Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-15T21:11:02.047108Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.047108Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:668e69bdbe7d9a2727b2761207ee34f363ca94b64305d72019f26b066304a8f1","observation_id":"a70cb5aa-ff92-409f-b95a-3405b51bbf27","resolution":{"observed_at":"2026-08-15T21:11:02.047108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.051703Z","title":"A primer in BERTology: What we know about how BERT works","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.051703Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:5caf7b8ae9e50809d195a5a8ec8b3ccfeee7b390c74c4c4d82b6cf87663f57ac","observation_id":"d5a5b1b9-7b0f-427c-973e-a741a2cd28cd","resolution":{"observed_at":"2026-08-15T21:11:02.051703Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.196936Z","title":"Research priorities for robust and beneficial artificial intelligence","venue":null,"work_id":"ca372a25-e789-4ef9-bfa3-87cbc0fdd2b1","year":2015},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.055985Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:c4a633b97ce2019d3457cb911695dbbd8665743bd53a734b68920781c30cfd4b","observation_id":"cb2fe07f-6a82-4fb7-bc5a-7b4cefa20efd","resolution":{"observed_at":"2026-08-15T21:11:03.201675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.180346Z","title":null,"venue":null,"work_id":"7a0f9880-038a-435d-9bc7-149393ee1d1e","year":2019},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.060734Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6d41bcd9e83c381754c2ade30c40f3df66b6f767fc8e2bd358d580d1b4e94979","observation_id":"7a2cf6c4-3cb3-42c1-a6c5-e33c3fd83511","resolution":{"observed_at":"2026-08-15T21:11:03.185336Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.04761","last_updated":"2023-02-09T16:49:57Z","snapshot_observed_at":"2026-08-22T05:04:27.685298Z","submitted_at":"2023-02-09T16:49:57Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.04761","snapshot_observed_at":"2026-08-15T21:11:02.064856Z","title":"Toolformer: Language models can teach themselves to use tools","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.064856Z"},"links":{"cited_paper":"/paper/2302.04761","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:16b2ea0b2319509fcbef501ccd0b8a7963ea03d683444acf78ff9614b0986054","observation_id":"95101970-bd83-42b3-bfc5-703b311591ce","resolution":{"observed_at":"2026-08-15T21:11:02.064856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.165755Z","title":"An evolutionary model of personality traits related to cooperative behavior using a large language model","venue":null,"work_id":"eeabdb4c-a419-4a55-ac68-df47d18dd88e","year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.069111Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ab03863ed3bd26d7fc9b0678092480666d6f24de9ae2f2a4e3b84d90cc272803","observation_id":"8f6167c9-4a5f-46af-b08b-065532a5c700","resolution":{"observed_at":"2026-08-15T21:11:03.170463Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s11948-022-00392-3","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-20T11:03:42.707129Z","title":"A comparative analysis of the definitions of autonomous weapons systems","venue":"Science and Engineering Ethics","work_id":"65909bda-98a2-459d-9909-612e9bd6f539","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.073714Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:fd31bc5e07e2099a466fe5d9eb058ec90e5cbfd08974bb3232d448e266116942","observation_id":"0b622b1d-44e4-43db-bc95-870dd9ca5525","resolution":{"observed_at":"2026-08-15T21:11:02.190554Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08295","last_updated":"2024-04-16T12:52:47Z","snapshot_observed_at":"2026-08-03T03:29:01.959523Z","submitted_at":"2024-03-13T06:59:16Z","title":"Gemma: Open Models Based on Gemini Research and Technology","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08295","snapshot_observed_at":"2026-08-15T21:11:02.078502Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.078502Z"},"links":{"cited_paper":"/paper/2403.08295","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:86e88706b1484969c07374ff790b8c6d16fbf14050f7ca827de2239fca14ad88","observation_id":"54cdc25e-dcbe-4a06-a3ca-a0b083ec66e7","resolution":{"observed_at":"2026-08-15T21:11:02.078502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-21T05:15:28.840279Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-15T21:11:02.082737Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.082737Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:4b0e3816bdf7f7cd0299d9100c98f4bf0dcf54ff396a9348f2467b0e8d0a1f06","observation_id":"e70192c0-2fa3-4d26-9e99-25bce4521ab8","resolution":{"observed_at":"2026-08-15T21:11:02.082737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:02.087316Z","title":"Scaling monosemanticity: Extracting interpretable features from claude 3 sonnet","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.087316Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ea5c6edb2137d516d6bfc794d205c6a4b930d94f831d90b0bb4d0f915f01e992","observation_id":"625f59c6-0eef-406d-9cf0-d8c92c8ef449","resolution":{"observed_at":"2026-08-15T21:11:02.087316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.138657Z","title":"Moral alignment for llm agents","venue":null,"work_id":"6eaaf476-88be-45c4-b4a1-e6404ca5891b","year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.092060Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:16a771f01da0fe7fbbc9c462e63e7df6c9927a2e5f2956862a50377636d585b6","observation_id":"7b2ebf77-d4d1-47eb-846b-c88fd4d56a1d","resolution":{"observed_at":"2026-08-15T21:11:03.143335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-15T21:11:02.098429Z","title":"Llama: Open and efficient foundation language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.098429Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:a73e10b37d6b130ee2da8cd30e069dab6ec3c3c1dc88da96c3e722df1727b64b","observation_id":"f523afad-184f-41ef-9a0e-57e19a11c506","resolution":{"observed_at":"2026-08-15T21:11:02.098429Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-15T21:11:02.104902Z","title":"Llama 2: Open foundation and fine-tuned chat models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.104902Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:488a0e92d73ee904c54b58e96d9ce7d2e3d963edcd6c20102a22f73679042d9b","observation_id":"82ac90b8-2d66-452e-9693-66c10e7d314b","resolution":{"observed_at":"2026-08-15T21:11:02.104902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-08-16T09:50:11.319379Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-08-15T21:11:02.109561Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.109561Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:84b8d05572d573c2a304019432bfdfcfd22929918023a6c927362076b26f2ec6","observation_id":"20ce1bc9-e4d8-4904-9888-9de2d34c5479","resolution":{"observed_at":"2026-08-15T21:11:02.109561Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.122837Z","title":"Mobile-agent: Autonomous multi-modal mobile device agent with visual perception,","venue":null,"work_id":"4964065f-8338-44e6-a00d-d8bbd9ae04dc","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.114293Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:f6ec9b91c481381947383deb822883dd291eca186605afe990e95c9578d21292","observation_id":"f2942ef8-188f-4e1f-a44b-0747ba8e3471","resolution":{"observed_at":"2026-08-15T21:11:03.127797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.107500Z","title":"Dai, and Quoc V Le","venue":null,"work_id":"2f54aaa0-ecf9-4fa9-a849-fb3d0f2f63ab","year":2022},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.123585Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:3d6f29712340fbd701b392da665a0487100d85568397e3183d7582e1c8e1f304","observation_id":"e8c5f846-e91b-401e-80a4-0fd5d73e1194","resolution":{"observed_at":"2026-08-15T21:11:03.112312Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-08-13T07:04:41.220509Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-15T21:11:02.128289Z","title":"Chain-of-thought prompting elicits reasoning in large language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.128289Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:e784b7c5d5b38e5c1bb39b91fedb824e072554b821f5b88bf7b8ceae8b7d13c0","observation_id":"9aac0e58-4508-4dce-b103-c860ac6c28f4","resolution":{"observed_at":"2026-08-15T21:11:02.128289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-15T21:11:02.133411Z","title":"ReAct: Synergizing reasoning and acting in language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.133411Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:12e109c5be4cd28bae23d50a0fc8ecd17221320eaa61ac2037adf5af4713c302","observation_id":"1cea8ac7-27e1-49cf-b37d-d6744ff11b68","resolution":{"observed_at":"2026-08-15T21:11:02.133411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13771","last_updated":"2026-07-05T07:51:04Z","snapshot_observed_at":"2026-08-20T18:18:30.378150Z","submitted_at":"2023-12-21T11:52:45Z","title":"AppAgent: Multimodal Agents as Smartphone Users","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13771","snapshot_observed_at":"2026-08-15T21:11:02.140648Z","title":"Appagent: Multimodal agents as smartphone users, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.140648Z"},"links":{"cited_paper":"/paper/2312.13771","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:ffc9db1b1edd25388cc4c38741306ed62b48171bd7b6d5b8014e0e3a6a763ac1","observation_id":"647ac5ed-565f-4692-8bee-b21bc1834a89","resolution":{"observed_at":"2026-08-15T21:11:02.140648Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16158","last_updated":"2024-04-18T06:53:38Z","snapshot_observed_at":"2026-08-16T13:42:20.592010Z","submitted_at":"2024-01-29T13:46:37Z","title":"Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.16158","snapshot_observed_at":"2026-08-15T21:11:02.118923Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.118923Z"},"links":{"cited_paper":"/paper/2401.16158","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:cb0932424bf65a1d6ab0532aa77530d229365fd2a50a934e3fd16b6fe282cfd0","observation_id":"33d30244-b423-4420-bcc9-fbd665b74b6d","resolution":{"observed_at":"2026-08-15T21:11:02.118923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01405","last_updated":"2025-03-03T06:14:14Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:07Z","title":"Representation Engineering: A Top-Down Approach to AI Transparency","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01405","snapshot_observed_at":"2026-08-15T21:11:02.149478Z","title":"Byun, Zifan Wang, Alex Mallen, Steven Basart, Sanmi Koyejo, Dawn Song, Matt Fredrikson, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.149478Z"},"links":{"cited_paper":"/paper/2310.01405","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:de3cd754c442b347905b0e5c04cce4a47adfa193e7f975aaf60c01be22ac6381","observation_id":"10bb76ee-bc2d-40f4-babf-40043502314b","resolution":{"observed_at":"2026-08-15T21:11:02.149478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11436","last_updated":"2024-06-07T04:52:29Z","snapshot_observed_at":"2026-08-20T03:55:01.596724Z","submitted_at":"2023-09-20T16:12:32Z","title":"You Only Look at Screens: Multimodal Chain-of-Action Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.11436","snapshot_observed_at":"2026-08-15T21:11:02.145319Z","title":"You only look at screens: Multimodal chain-of-action agents, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:02.145319Z"},"links":{"cited_paper":"/paper/2309.11436","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:013fe3a399542f12e0a6157c214bcbd433d7ca25a11f9c77b1104456f200e958","observation_id":"e1633c79-0375-4295-bfa3-8325c644a49a","resolution":{"observed_at":"2026-08-15T21:11:02.145319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:01.984893Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.984893Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:4e20b00e5435251add6c22a789920ee7cb52bc422b38f2720cc1dc7d32a4c7d8","observation_id":"d843c580-4512-4d8f-a3a6-6b41724a5ec6","resolution":{"observed_at":"2026-08-15T21:11:01.984893Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03874","last_updated":"2021-11-08T21:30:18Z","snapshot_observed_at":"2026-08-21T04:25:41.592030Z","submitted_at":"2021-03-05T18:59:39Z","title":"Measuring Mathematical Problem Solving With the MATH Dataset","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03874","snapshot_observed_at":"2026-08-15T21:11:01.925637Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.925637Z"},"links":{"cited_paper":"/paper/2103.03874","citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:2a1027b9de6115620f3840d9b0dbf3b02318c94e1af4425a1480369fa540ce33","observation_id":"671ed4ec-d1ad-4d99-97f8-e0d15322b52b","resolution":{"observed_at":"2026-08-15T21:11:01.925637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.469124Z","title":null,"venue":null,"work_id":"9cbe17c8-0558-45a5-81b9-224ec7f22f69","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.866455Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:8e69029f32e399eb51a36d05ae4c1e5eb85f3183a8d8122161f860c00d3d6a9a","observation_id":"6b1e14ac-f124-489b-84f4-00a88d7ea70d","resolution":{"observed_at":"2026-08-15T21:11:03.474249Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T21:11:03.271093Z","title":null,"venue":null,"work_id":"9b87a092-16aa-452f-b667-1928c6c93cf8","year":null},"citing_paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T21:11:01.998638Z"},"links":{"citing_paper":"/paper/2505.10670"},"observation_digest":"sha256:6086029533cde48ae5ad2669dc83a02a90b94ea6b2a4e6deb0950417257bb98b","observation_id":"98fbd528-7341-4dfe-92d5-440307b35277","resolution":{"observed_at":"2026-08-15T21:11:03.276283Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.10670","last_updated":"2025-05-15T19:22:11Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-20T07:22:13.070506Z","submitted_at":"2025-05-15T19:22:11Z","title":"Interpretable Risk Mitigation in LLM Agent Systems"},"reference_resolution":{"displayed":77,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":49,"verified_exact":2,"verified_fuzzy":24},"total_outbound_references":77},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 77 of 77 outbound references and 3 inbound Pith citation observations for arXiv:2505.10670."}