{"as_of":"2026-08-09T03:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:03606be80e974feb2b2460ac81aee4575e1b48bfb3c3c5a07dbc2a4e0df1ae1b","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T23:52:53.822416Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.15218/citation-record","integrity":"/paper/2607.15218/integrity","json":"/paper/2607.15218/citation-record.json","paper":"/paper/2607.15218"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-01T23:52:50.942964Z","title":"Phi-3 technical re- port: A highly capable language model locally on your phone.arXiv preprint arXiv:2404.14219,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:50.942964Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:752c1c6a24fadbe9bf04a302ac7fc646f8f91b8deed110736c09892413d408b8","observation_id":"1fc52820-4435-49d1-be5c-a05ae20aca2e","resolution":{"observed_at":"2026-08-01T23:52:50.942964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-01T23:52:51.152417Z","title":"Constitutional AI: Harmlessness from AI feedback.arXiv preprint arXiv:2212.08073,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.152417Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:2a308f78eda596de48340d52fefdb914e41e029ed67eb5379942c2480201d9f7","observation_id":"99c3a7f3-b077-4896-a13a-3bcc5206d674","resolution":{"observed_at":"2026-08-01T23:52:51.152417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02737","last_updated":"2025-02-04T21:43:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-04T21:43:16Z","title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02737","snapshot_observed_at":"2026-08-01T23:52:51.225771Z","title":"SmolLM2: When smol goes big – data-centric training of a small language model.arXiv preprint arXiv:2502.02737,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.225771Z"},"links":{"cited_paper":"/paper/2502.02737","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:0d1ef6de0e0a08f9b18411de2349a1db597c27f221dc7dc4e564544f69426f4b","observation_id":"8e25b957-f304-4cbb-9607-cd4b12107e11","resolution":{"observed_at":"2026-08-01T23:52:51.225771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.548149Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.548149Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:58e84f08bbbe5854cff368e325b9f4933011f77b6ec4670da218548a6c6329aa","observation_id":"3c92058b-21ed-4451-9603-3beaa4b563dd","resolution":{"observed_at":"2026-08-01T23:52:51.548149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05993","last_updated":"2024-09-11T14:42:29Z","snapshot_observed_at":"2026-08-06T16:17:24.580021Z","submitted_at":"2024-04-09T03:54:28Z","title":"AEGIS: Online Adaptive AI Content Safety Moderation with Ensemble of LLM Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05993","snapshot_observed_at":"2026-08-01T23:52:51.607980Z","title":"AEGIS: On- line adaptive AI content safety moderation with ensemble of LLM experts.arXiv preprint arXiv:2404.05993,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.607980Z"},"links":{"cited_paper":"/paper/2404.05993","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:db8b65c9ec5ce6590296e58c63aad0167fb905101aaadeddd8a8f2fe1762d003","observation_id":"0a908996-fc3e-49fb-9ad0-651b6a13bb62","resolution":{"observed_at":"2026-08-01T23:52:51.607980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14650","last_updated":"2025-04-20T15:12:14Z","snapshot_observed_at":"2026-08-07T16:00:46.123625Z","submitted_at":"2025-04-20T15:12:14Z","title":"A Framework for Benchmarking and Aligning Task-Planning Safety in LLM-Based Embodied Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14650","snapshot_observed_at":"2026-08-01T23:52:51.679091Z","title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.679091Z"},"links":{"cited_paper":"/paper/2504.14650","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:335ebd36515f48bbc8ae91c75a3fa6ca6eaaa21cd690d945c90d6d42cc13010e","observation_id":"3e103e25-3104-4283-b76d-e040d337b233","resolution":{"observed_at":"2026-08-01T23:52:51.679091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06674","last_updated":"2023-12-07T19:40:50Z","snapshot_observed_at":"2026-07-06T17:00:00.321552Z","submitted_at":"2023-12-07T19:40:50Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06674","snapshot_observed_at":"2026-08-01T23:52:51.813389Z","title":"Llama guard: LLM- based input-output safeguard for human-AI conversations.arXiv preprint arXiv:2312.06674,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.813389Z"},"links":{"cited_paper":"/paper/2312.06674","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:fd0a14c281f65087f63bc8164a44b0724dabc7b40b73c363e322fdff7af9ad08","observation_id":"b9897fd6-cab1-43d9-af9a-a2a9f498b85d","resolution":{"observed_at":"2026-08-01T23:52:51.813389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05474","last_updated":"2022-08-26T17:12:17Z","snapshot_observed_at":"2026-07-06T06:14:28.435222Z","submitted_at":"2017-12-14T23:17:24Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.05474","snapshot_observed_at":"2026-08-01T23:52:51.884733Z","title":"AI2-THOR: An interactive 3D environment for visual AI.arXiv preprint arXiv:1712.05474,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.884733Z"},"links":{"cited_paper":"/paper/1712.05474","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:12b76dd691c286018d3f14028618117be915cf443ea1d11806907f2384273bab","observation_id":"f307715a-0c52-4172-bcb3-d55374f2fb78","resolution":{"observed_at":"2026-08-01T23:52:51.884733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.980971Z","title":"AGENTSAFE: Benchmarking the safety of embodied agents on hazardous instructions.arXiv preprint arXiv:2506.14697,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.980971Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:b7d967106c940b16837ee93164920445001b699bb495bd0d2282aec3df85f835","observation_id":"45d32663-42a5-4dc4-af10-23bdb3dd896e","resolution":{"observed_at":"2026-08-01T23:52:51.980971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:52.049845Z","title":"G-Eval: NLG evaluation using GPT-4 with better human alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.049845Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:c09b05dc35eb685573e6cfa1bcea261ed1a07180d1d23397f978b0d35ebb97ca","observation_id":"04633014-3ac9-4779-ad91-b89bf0da1569","resolution":{"observed_at":"2026-08-01T23:52:52.049845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-01T23:52:52.190795Z","title":"The llama 3 herd of models.arXiv preprint arXiv:2407.21783,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.190795Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:8dd978f325a38f9f3aad5606a462b7c7e02bd957144eff12a41b7316502ecd58","observation_id":"99da4540-4375-4875-b54d-c7697ac349a4","resolution":{"observed_at":"2026-08-01T23:52:52.190795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:52.274400Z","title":"IS-Bench: Evaluating interactive safety of VLM-driven embodied agents in daily household tasks.arXiv preprint arXiv:2506.16402,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.274400Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:7dd7ea9ab5c1005e200f26a60a396079ff99b795c5e0d64a96da8d410cc6853a","observation_id":"e51a3e38-95ef-4544-b86e-85866116c1f4","resolution":{"observed_at":"2026-08-01T23:52:52.274400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06824","last_updated":"2024-08-19T01:18:41Z","snapshot_observed_at":"2026-07-06T16:30:37.867641Z","submitted_at":"2023-10-10T17:54:39Z","title":"The Geometry of Truth: Emergent Linear Structure in Large Language Model Representations of True/False Datasets","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06824","snapshot_observed_at":"2026-08-01T23:52:52.384452Z","title":"The geometry of truth: Emergent linear structure in LLM repre- sentations of true/false datasets.arXiv preprint arXiv:2310.06824,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.384452Z"},"links":{"cited_paper":"/paper/2310.06824","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:24378aef1b8f4988de4f54acfe1cf097f611fba2bd7a89ec6dfcc0986978f9ad","observation_id":"3da758d5-4aee-449c-819c-4ebfcceacde1","resolution":{"observed_at":"2026-08-01T23:52:52.384452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:52.538702Z","title":"Don’t let your robot be harmful: Responsible robotic manipulation via safety- as-policy.arXiv preprint arXiv:2411.18289,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.538702Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:192810ea036e6a817d7cf9fc356ecf83ae1323b5afdf310a563a6c2cec57884e","observation_id":"5683ea90-6138-47ea-bb6e-be781e9317ad","resolution":{"observed_at":"2026-08-01T23:52:52.538702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06892","last_updated":"2025-03-10T03:37:36Z","snapshot_observed_at":"2026-08-07T17:18:06.193061Z","submitted_at":"2025-03-10T03:37:36Z","title":"SafePlan: Leveraging Formal Logic and Chain-of-Thought Reasoning for Enhanced Safety in LLM-based Robotic Task Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06892","snapshot_observed_at":"2026-08-01T23:52:52.704841Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.704841Z"},"links":{"cited_paper":"/paper/2503.06892","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:8da981d831395a784d491ee7a8a2b243097992b1a26c75e0f5a9e8c845274aa9","observation_id":"22e84c45-dd55-4ab4-baf2-a22da5c9bca9","resolution":{"observed_at":"2026-08-01T23:52:52.704841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-01T23:52:52.872352Z","title":"GPT-4 technical report.arXiv preprint arXiv:2303.08774,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.872352Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:969d272b52c4b6d6482bc8bf4a6d62d27de3966f897c5f354ecca03330c8aeea","observation_id":"24fd2da0-bf5b-4d9c-87f0-2bea544254ca","resolution":{"observed_at":"2026-08-01T23:52:52.872352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-01T23:52:53.302816Z","title":"Llama 2: Open founda- tion and fine-tuned chat models.arXiv preprint arXiv:2307.09288,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.302816Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:dc47ac00e59648a7f1fe09f5f6bc1b8c3eeb14d7df5ad462bb51a0c70ce8ff21","observation_id":"d98b5c67-da3e-4355-bc3b-be410e503ddb","resolution":{"observed_at":"2026-08-01T23:52:53.302816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10248","last_updated":"2024-10-10T13:20:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-20T12:21:05Z","title":"Steering Language Models With Activation Engineering","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10248","snapshot_observed_at":"2026-08-01T23:52:53.373741Z","title":"Vazquez, Ulisse Mini, and Monte MacDiarmid","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.373741Z"},"links":{"cited_paper":"/paper/2308.10248","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:2be4fe247764a03e31856114ee494efe0a2f16b40a8e590c41902d72f931f111","observation_id":"5ec62ffe-87e2-4754-ba2d-9e91bed1400e","resolution":{"observed_at":"2026-08-01T23:52:53.373741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16291","last_updated":"2023-10-19T16:27:03Z","snapshot_observed_at":"2026-08-07T08:29:46.650400Z","submitted_at":"2023-05-25T17:46:38Z","title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16291","snapshot_observed_at":"2026-08-01T23:52:53.449275Z","title":"V oyager: An open-ended embodied agent with large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.449275Z"},"links":{"cited_paper":"/paper/2305.16291","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:67f925d74706c3388c925502f1c0b843340953b6235057f5d38400b3c7b5e1d5","observation_id":"1e1e2cb9-250e-4926-b811-27d758c4e5ab","resolution":{"observed_at":"2026-08-01T23:52:53.449275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15699","last_updated":"2025-06-19T04:21:00Z","snapshot_observed_at":"2026-08-07T16:00:19.134747Z","submitted_at":"2025-04-22T08:34:35Z","title":"Advancing Embodied Agent Security: From Safety Benchmarks to Input Moderation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15699","snapshot_observed_at":"2026-08-01T23:52:53.514874Z","title":"Advancing embodied agent security: From safety benchmarks to input moderation.arXiv preprint arXiv:2504.15699,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.514874Z"},"links":{"cited_paper":"/paper/2504.15699","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:89e7637ad12428df953d289bd3a9450d915aa5fc50a8fb890b574fdf00877ad9","observation_id":"b8684d64-8159-4a6d-a6ac-07c8750107e7","resolution":{"observed_at":"2026-08-01T23:52:53.514874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:53.574727Z","title":"SafeAgentBench: A benchmark for safe task planning of embodied LLM agents.arXiv preprint arXiv:2412.13178,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.574727Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:dd271772959756e90f092e356c53213956dec96952cc58152b9bcbc219c31cdc","observation_id":"490fb07f-1faa-4fbf-abf3-7c9a68b1cd51","resolution":{"observed_at":"2026-08-01T23:52:53.574727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:53.629239Z","title":"R-Judge: Bench- marking safety risk awareness for LLM agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.629239Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:97658bfbdbe954d8339b70cbfbd147b90ddb2b1215ad18055909fcbbafc35213","observation_id":"b77de546-c204-434f-b5c5-7b7d79ddacb7","resolution":{"observed_at":"2026-08-01T23:52:53.629239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21772","snapshot_observed_at":"2026-08-01T23:52:53.689569Z","title":"ShieldGemma: Generative AI content moderation based on gemma.arXiv preprint arXiv:2407.21772,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.689569Z"},"links":{"cited_paper":"/paper/2407.21772","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:4ce1a2d5148408bd1077771cc0c2491fac18395ef81114617a0a62e7a0ec0b8c","observation_id":"bf547adc-49f9-4247-a269-2ae631839471","resolution":{"observed_at":"2026-08-01T23:52:53.689569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04449","last_updated":"2024-11-28T12:28:02Z","snapshot_observed_at":"2026-07-06T18:58:22.457329Z","submitted_at":"2024-08-08T13:19:37Z","title":"EARBench: Towards Evaluating Physical Risk Awareness for Task Planning of Foundation Model-based Embodied AI Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04449","snapshot_observed_at":"2026-08-01T23:52:53.743894Z","title":"EARBench: Towards evaluating physical risk awareness for task planning of foundation model-based embod- ied AI agents.arXiv preprint arXiv:2408.04449,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.743894Z"},"links":{"cited_paper":"/paper/2408.04449","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:38bf340ee500c51bef2b6060b1d8b4132348c9704df7d246b32d83814a495bbe","observation_id":"02dd3a31-fca3-4fbf-8a7b-fb2e2aa900c6","resolution":{"observed_at":"2026-08-01T23:52:53.743894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01405","last_updated":"2025-03-03T06:14:14Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:07Z","title":"Representation Engineering: A Top-Down Approach to AI Transparency","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01405","snapshot_observed_at":"2026-08-01T23:52:53.822416Z","title":"SDD”/“PGD","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.822416Z"},"links":{"cited_paper":"/paper/2310.01405","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:2221130caf9ed2f222ac2f58e11e168d096554c9eb7406dc3e0d1d5eed8a9f4e","observation_id":"1b5897e3-f3dd-407e-934d-6791ca3f3427","resolution":{"observed_at":"2026-08-01T23:52:53.822416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.934761Z","title":"SafeText: A benchmark for exploring physical safety in language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.934761Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:0e5dd1eea481994d07c6652b5166457aaefc0e5eb19898a35ae5c0d572ddcc7a","observation_id":"d07c7891-e54a-4e97-803f-9b342e3d8767","resolution":{"observed_at":"2026-08-01T23:52:51.934761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-01T23:52:53.025835Z","title":"Qwen2.5 technical report.arXiv preprint arXiv:2412.15115,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.025835Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:0aaa35cfa8873f93b858aecb525c4d4d3f98b5763dd8f26885b343bd35bf0c86","observation_id":"45d71624-eabe-449b-aca1-166d013169a5","resolution":{"observed_at":"2026-08-01T23:52:53.025835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.08663","last_updated":"2025-03-11T17:50:47Z","snapshot_observed_at":"2026-08-07T17:12:15.755345Z","submitted_at":"2025-03-11T17:50:47Z","title":"Generating Robot Constitutions & Benchmarks for Semantic Safety","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.08663","snapshot_observed_at":"2026-08-01T23:52:53.215213Z","title":"Gen- erating robot constitutions & benchmarks for semantic safety.arXiv preprint arXiv:2503.08663,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.215213Z"},"links":{"cited_paper":"/paper/2503.08663","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:b0409b644fb82f30ba06947e2ab8e5ba73708c1d65e8e9ce65ca4eb6dfa0b9cc","observation_id":"7929156b-5393-4500-9fa6-acfdce7b5cea","resolution":{"observed_at":"2026-08-01T23:52:53.215213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-01T23:52:51.313791Z","title":"RT-2: Vision-language-action models transfer web knowledge to robotic control.arXiv preprint arXiv:2307.15818,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.313791Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:20e672c7f9a72c2d10343e10056c614fff398a9298bd1ed4643419dc9da3fcbe","observation_id":"d6268bef-04c6-4401-a31a-56d243b830a0","resolution":{"observed_at":"2026-08-01T23:52:51.313791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1610.01644","last_updated":"2018-11-22T23:40:00Z","snapshot_observed_at":"2026-07-06T05:13:30.860932Z","submitted_at":"2016-10-05T20:59:01Z","title":"Understanding intermediate layers using linear classifier probes","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1610.01644","snapshot_observed_at":"2026-08-01T23:52:51.078723Z","title":"Understanding intermediate layers using linear classifier probes.arXiv preprint arXiv:1610.01644,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.078723Z"},"links":{"cited_paper":"/paper/1610.01644","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:14113c09aca9d25a9007ac0c99c110394648c1cf0115952e7803d781e30035e0","observation_id":"1af75e5c-fa88-41ee-8bc1-d41f74c9ba51","resolution":{"observed_at":"2026-08-01T23:52:51.078723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.422778Z","title":"SafeMind: Benchmarking and mitigating safety risks in embodied LLM agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.422778Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:c200fda9e2edaa5d074fb8b6cb85365fd9ddf948654368dd8e692a31e9756412","observation_id":"6ee6a447-c712-4375-98d0-1a3d8f75ca07","resolution":{"observed_at":"2026-08-01T23:52:51.422778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.01691","last_updated":"2022-08-16T16:06:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-04T17:57:11Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.01691","snapshot_observed_at":"2026-08-01T23:52:51.006565Z","title":"Do as I can, not as I say: Grounding language in robotic affordances.arXiv preprint arXiv:2204.01691,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.006565Z"},"links":{"cited_paper":"/paper/2204.01691","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:efdd74b305087f3577d2a4a7797eaf3d53ee9e48fafd00800a9299fd0c9970f3","observation_id":"01797a0e-f032-45d7-a546-b4681acb2661","resolution":{"observed_at":"2026-08-01T23:52:51.006565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16174","last_updated":"2026-06-01T15:28:47Z","snapshot_observed_at":"2026-08-07T17:56:39.289394Z","submitted_at":"2025-02-22T10:31:50Z","title":"Efficient LLM Moderation with Multi-Layer Latent Prototypes","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.16174","snapshot_observed_at":"2026-08-01T23:52:51.468348Z","title":"Efficient LLM moderation with multi-layer latent prototypes.arXiv preprint arXiv:2502.16174,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.468348Z"},"links":{"cited_paper":"/paper/2502.16174","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:d11ef67c45e9d8cff2b09ae5aaa88a7b8ed2b6c6f7b8a43d45a576ef80e43d60","observation_id":"520e3158-13b0-453c-9911-a67b79e05fe0","resolution":{"observed_at":"2026-08-01T23:52:51.468348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-08T03:08:00.506171Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":33,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 0 inbound Pith citation observations for arXiv:2607.15218."}