{"as_of":"2026-08-10T11:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8365e5ccb4a75ca4af6a1312b3b6ea4360a2a553bb19625702f087e01b370058","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":44,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-09T13:45:55.122362Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":18,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2402.19173","last_updated":"2024-02-29T13:53:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-29T13:53:35Z","title":"StarCoder 2 and The Stack v2: The Next Generation","version":1},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-05-12T17:28:22.353355Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2402.19173"},"observation_digest":"sha256:7cb247a7056131f9c43cb5ed2972941672afaf8de9ee58dd5c4544eed30be573","observation_id":"d71ed716-aa49-42ae-a6c4-491f44c60ad0","resolution":{"observed_at":"2026-05-12T17:28:22.675474Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2411.10656","last_updated":"2026-04-03T16:24:12Z","snapshot_observed_at":"2026-07-06T19:51:14.080262Z","submitted_at":"2024-11-16T01:31:29Z","title":"Precision or Peril: A PoC of Python Code Quality from Quantized Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T17:30:54.204300Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2411.10656"},"observation_digest":"sha256:a2e83cbcf81494548aa3bce776f900f534fe942a948b6e0e03f72d399e0fd280","observation_id":"4299a779-96e4-4419-9105-def9b99382a0","resolution":{"observed_at":"2026-05-23T17:33:14.955539Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T18:40:50.139345Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2501.14249"},"observation_digest":"sha256:ae11003edd3fa3b1ed214ade1a33dae9556261908792f895500f05db019d2d68","observation_id":"f303afb9-da19-4460-9313-d870f3e455bd","resolution":{"observed_at":"2026-05-10T18:40:50.368040Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-09T13:45:55.122362Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.02009","last_updated":"2025-02-04T04:56:34Z","snapshot_observed_at":"2026-08-09T23:28:28.808011Z","submitted_at":"2025-02-04T04:56:34Z","title":"LLMSecConfig: An LLM-Based Approach for Fixing Software Container Misconfigurations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-09T13:45:55.122362Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2502.02009"},"observation_digest":"sha256:048c850dfed74013029360be8993add0e5377706b05f905a940307ee0b021c35","observation_id":"301c7b7c-519f-472b-9145-8a07f717f1db","resolution":{"observed_at":"2026-08-09T13:45:55.122362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-08T17:00:30.464049Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06039","last_updated":"2025-02-09T21:23:07Z","snapshot_observed_at":"2026-08-09T23:27:31.623998Z","submitted_at":"2025-02-09T21:23:07Z","title":"Benchmarking Prompt Engineering Techniques for Secure Code Generation with GPT Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T17:00:30.464049Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2502.06039"},"observation_digest":"sha256:531b5976e7c107b5139c2da2f372d738183a7eae48f2a124d1f559c2a0b239de","observation_id":"3e89f275-13dc-49a2-8358-52d8b34ab118","resolution":{"observed_at":"2026-08-08T17:00:30.464049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T13:04:58.231629Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.22704","last_updated":"2025-05-28T17:57:47Z","snapshot_observed_at":"2026-08-09T22:36:48.949951Z","submitted_at":"2025-05-28T17:57:47Z","title":"Training Language Models to Generate Quality Code with Program Analysis Feedback","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T13:04:58.231629Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2505.22704"},"observation_digest":"sha256:e0755663e759390452b6dfad932a5e8cbd7673317f4df99a94df71b804030071","observation_id":"259fba11-9ee5-4183-a195-facc5c07f03b","resolution":{"observed_at":"2026-08-07T13:04:58.231629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T12:46:10.267103Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.23634","last_updated":"2025-05-29T16:44:29Z","snapshot_observed_at":"2026-08-09T15:33:48.720032Z","submitted_at":"2025-05-29T16:44:29Z","title":"MCP Safety Training: Learning to Refuse Falsely Benign MCP Exploits using Improved Preference Alignment","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:46:10.267103Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2505.23634"},"observation_digest":"sha256:ce633f15478e5d2d6946fbeac3bec2a74f57d244324849c7581653986b0c478a","observation_id":"6f606523-bef9-46e5-b411-10bf0bb1bb23","resolution":{"observed_at":"2026-08-07T12:46:10.267103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T11:52:57.988743Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02066","last_updated":"2025-06-01T23:37:41Z","snapshot_observed_at":"2026-08-09T07:08:16.872886Z","submitted_at":"2025-06-01T23:37:41Z","title":"Developing a Risk Identification Framework for Foundation Model Uses","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:52:57.988743Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.02066"},"observation_digest":"sha256:c7c28001d94eca2ef0572cbfc5a82798e055ae86c37eccebd0b274e3e6d45a14","observation_id":"2a740e82-3223-4dfb-9470-ccff15569fbc","resolution":{"observed_at":"2026-08-07T11:52:57.988743Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T05:42:54.911928Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.07313","last_updated":"2025-06-08T23:08:08Z","snapshot_observed_at":"2026-08-10T11:09:11.672263Z","submitted_at":"2025-06-08T23:08:08Z","title":"SCGAgent: Recreating the Benefits of Reasoning Models for Secure Code Generation with Agentic Workflows","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T05:42:54.911928Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.07313"},"observation_digest":"sha256:dda78f2c68bc26daff9c7ef9c35f897de4bb00c69ba318934d8a4a4e73f776a8","observation_id":"9e5f17a6-a27d-47b9-b7eb-835e73519e26","resolution":{"observed_at":"2026-08-07T05:42:54.911928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-07T10:17:26.763796Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.763796Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:d31e0588dde42f85412f9f71fa86342d93874491c80b941054d521068b809fa8","observation_id":"78188daa-cfd8-411e-b556-0bf09ef0c4f5","resolution":{"observed_at":"2026-08-07T10:17:26.763796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T21:56:13.509418Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23034","last_updated":"2025-06-28T23:24:33Z","snapshot_observed_at":"2026-08-08T22:22:43.604466Z","submitted_at":"2025-06-28T23:24:33Z","title":"Guiding AI to Fix Its Own Flaws: An Empirical Study on LLM-Driven Secure Code Generation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:56:13.509418Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2506.23034"},"observation_digest":"sha256:077df013760ff14c603e511c66d2f34b67b22a5f5c339d26f6a3ccdabadafa24","observation_id":"381c30dd-e45d-4868-af24-85cbb8ecd9f9","resolution":{"observed_at":"2026-08-06T21:56:13.509418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T20:45:53.247909Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02057","last_updated":"2025-07-02T18:00:49Z","snapshot_observed_at":"2026-08-10T11:19:13.970097Z","submitted_at":"2025-07-02T18:00:49Z","title":"MGC: A Compiler Framework Exploiting Compositional Blindness in Aligned LLMs for Malware Generation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:45:53.247909Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2507.02057"},"observation_digest":"sha256:4ec3f37374e0c283c570f4789bb91fd4cf8c8a0371213dec59b46ea2e186c45d","observation_id":"cf6837e0-a1cf-456c-b41d-df3c068a6907","resolution":{"observed_at":"2026-08-06T20:45:53.247909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T14:45:42.847281Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models.arXiv preprint arXiv:2312.04724,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18105","last_updated":"2025-07-24T05:30:54Z","snapshot_observed_at":"2026-08-07T21:22:02.859315Z","submitted_at":"2025-07-24T05:30:54Z","title":"Understanding the Supply Chain and Risks of Large Language Model Applications","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:42.847281Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2507.18105"},"observation_digest":"sha256:856ca80d0b2ffd9503be6135a9406cf5cbd37785e4d4d58f3a4b85a56396d50e","observation_id":"97794215-ada6-49f4-9f3e-40a02bb0b665","resolution":{"observed_at":"2026-08-06T14:45:42.847281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T14:25:15.757169Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19399","last_updated":"2025-07-25T16:06:16Z","snapshot_observed_at":"2026-08-09T23:41:06.417558Z","submitted_at":"2025-07-25T16:06:16Z","title":"Running in CIRCLE? A Simple Benchmark for LLM Code Interpreter Security","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T14:25:15.757169Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2507.19399"},"observation_digest":"sha256:0033bcd3caf35fc59cffa5d78d420977dd72a1bc96f071d6528e14b5a896cb2a","observation_id":"5dfe7ef9-3eb8-4fa5-8515-d47d7e71faec","resolution":{"observed_at":"2026-08-06T14:25:15.757169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T01:03:49.483750Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.03933","last_updated":"2025-08-05T21:55:14Z","snapshot_observed_at":"2026-08-10T07:07:18.397746Z","submitted_at":"2025-08-05T21:55:14Z","title":"Towards terahertz nanomechanics","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T01:03:49.483750Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2508.03933"},"observation_digest":"sha256:4d22379b80a32067f16a4504b57d149ec06b18a94d7eb9372e94253a92a1d1ee","observation_id":"24a861b7-a4e2-4ef7-8c4e-c1211673c999","resolution":{"observed_at":"2026-08-06T01:03:49.483750Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-06T01:04:32.435944Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.03936","last_updated":"2025-08-05T21:57:52Z","snapshot_observed_at":"2026-08-08T12:25:04.397522Z","submitted_at":"2025-08-05T21:57:52Z","title":"ASTRA: Autonomous Spatial-Temporal Red-teaming for AI Software Assistants","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T01:04:32.435944Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2508.03936"},"observation_digest":"sha256:ce5a4fde4f27a1211492063ec2e3199c6ce654f941a625c43e5f1ee5da7612fa","observation_id":"75bcb548-44e5-45d4-8578-254e6d951780","resolution":{"observed_at":"2026-08-06T01:04:32.435944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-03T23:50:49.056328Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.03898","last_updated":"2025-11-05T22:46:24Z","snapshot_observed_at":"2026-08-07T13:07:19.220144Z","submitted_at":"2025-11-05T22:46:24Z","title":"Secure Code Generation at Scale with Reflexion","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T23:50:49.056328Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2511.03898"},"observation_digest":"sha256:2041b367dd26b93a5bc23081f6d80aca2155d35ddad032fe5f8449f797b276ed","observation_id":"1f80b99b-ddea-4efa-bee6-62a473fe1d3f","resolution":{"observed_at":"2026-08-03T23:50:49.056328Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2512.05439","last_updated":"2026-05-07T21:46:58Z","snapshot_observed_at":"2026-08-06T07:24:27.198375Z","submitted_at":"2025-12-05T05:34:06Z","title":"BEAVER: An Efficient Deterministic LLM Verifier","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T01:58:44.719715Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2512.05439"},"observation_digest":"sha256:faba9b8bc5b21ff1afd134d9a9f299f29be496c60f9fe68f9b598d49e78a62ad","observation_id":"375b5ec1-aada-443f-8522-c7944673692a","resolution":{"observed_at":"2026-05-17T01:58:51.237959Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-03T05:18:42.953661Z","title":"Cao, X., Jia, J., and Gong, N","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2602.04894","last_updated":"2026-06-05T16:30:46Z","snapshot_observed_at":"2026-08-06T23:30:53.886306Z","submitted_at":"2026-02-02T22:23:36Z","title":"Extracting Recurring Vulnerabilities from Black-Box LLM-Generated Software","version":4},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T05:18:42.953661Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2602.04894"},"observation_digest":"sha256:77a9c4210ec9aa4ea92b7e2021003c7267ece5edbd07aedd3c95780e7ac9e42d","observation_id":"12076545-9314-47ab-8e36-96d721f8ed1e","resolution":{"observed_at":"2026-08-03T05:18:42.953661Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2602.06759","last_updated":"2026-05-14T07:04:32Z","snapshot_observed_at":"2026-08-04T02:39:11.494645Z","submitted_at":"2026-02-06T15:06:36Z","title":"\"Tab, Tab, Bug\": Security Pitfalls of Next Edit Suggestions in AI-Integrated IDEs","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T06:53:13.235588Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2602.06759"},"observation_digest":"sha256:47287881364c0d6760e10edde2e4382827c6cd68bc7d53155496ba54ae0fee99","observation_id":"1c6d0de2-b199-4885-869e-e2348c008998","resolution":{"observed_at":"2026-05-16T06:57:29.297748Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.05292","last_updated":"2026-04-08T16:49:43Z","snapshot_observed_at":"2026-07-06T22:54:04.491386Z","submitted_at":"2026-04-07T00:55:42Z","title":"Broken by Default: A Formal Verification Study of Security Vulnerabilities in AI-Generated Code","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T20:12:03.118074Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.05292"},"observation_digest":"sha256:3e949a127b1cad585f975196c4a53576dc804a171329eb4ebe3d63aa1274a90e","observation_id":"94d15de5-8ac2-4d03-9fd8-6d3157135cbb","resolution":{"observed_at":"2026-05-10T22:10:48.566036Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.09544","last_updated":"2026-07-03T15:37:04Z","snapshot_observed_at":"2026-08-01T18:28:17.308608Z","submitted_at":"2026-04-10T17:58:31Z","title":"Large Language Models Generate Harmful Responses Using a Distinct Mechanism, Shared Across Harm Types","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T17:08:25.471462Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.09544"},"observation_digest":"sha256:a7e24595b90593d51c0eb4a3f41983c8df38a617751662758588187afccabc17","observation_id":"bebf90b3-ee10-42c9-9deb-daf54a51f640","resolution":{"observed_at":"2026-05-11T07:31:00.338647Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.17803","last_updated":"2026-04-20T04:51:39Z","snapshot_observed_at":"2026-08-02T12:38:39.791138Z","submitted_at":"2026-04-20T04:51:39Z","title":"Adversarial Arena: Crowdsourcing Data Generation through Interactive Competition","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-10T04:55:43.987116Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.17803"},"observation_digest":"sha256:4ef3dff6164682d5254d19289105649d07aa590adfdf8ac6caad96c9f2157824","observation_id":"7fbbb1e4-aa8e-49f4-b162-b551afa5aa3e","resolution":{"observed_at":"2026-05-10T11:10:09.437088Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2604.18718","last_updated":"2026-04-20T18:17:51Z","snapshot_observed_at":"2026-08-05T14:04:49.062431Z","submitted_at":"2026-04-20T18:17:51Z","title":"Towards Optimal Agentic Architectures for Offensive Security Tasks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T04:02:04.359269Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2604.18718"},"observation_digest":"sha256:b6ef1edd04e28ddfb83840dfb1037d371994e4d397dbb50b0280293a6b3c4f7b","observation_id":"2c5d9f52-8c89-4d6e-9660-35e553710ce1","resolution":{"observed_at":"2026-05-11T12:16:03.146927Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.03179","last_updated":"2026-05-04T21:42:10Z","snapshot_observed_at":"2026-08-05T14:29:07.431614Z","submitted_at":"2026-05-04T21:42:10Z","title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-08T18:11:29.066362Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.03179"},"observation_digest":"sha256:07a8faec718a61c6a83a313e0c81889d3ee086b71ae737cad912fe727fd51041","observation_id":"3eda2f9c-237c-4b36-803e-1bdc8981c0d5","resolution":{"observed_at":"2026-05-09T06:40:43.797249Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.04019","last_updated":"2026-05-05T17:43:52Z","snapshot_observed_at":"2026-07-06T23:16:49.553583Z","submitted_at":"2026-05-05T17:43:52Z","title":"Redefining AI Red Teaming in the Agentic Era: From Weeks to Hours","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-07T16:06:18.057868Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.04019"},"observation_digest":"sha256:a3f7f4b050941e7ed6974bdc7bd400a6eaefd856afa46c4a8b4e42397b04b85c","observation_id":"89af28ed-bcf5-40d3-8611-5630b43ad2ef","resolution":{"observed_at":"2026-05-11T23:51:44.776533Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.05267","last_updated":"2026-05-06T09:38:31Z","snapshot_observed_at":"2026-08-04T17:41:12.164736Z","submitted_at":"2026-05-06T09:38:31Z","title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-08T17:37:51.790000Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.05267"},"observation_digest":"sha256:be07d98ba8f83e124ff3ec71b92ee96db8a8b1f9e611576eacb963a2d146a86b","observation_id":"15bad8b2-2d34-44a3-928e-b4a531ea6e94","resolution":{"observed_at":"2026-05-11T17:26:04.477648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.08382","last_updated":"2026-05-08T18:40:47Z","snapshot_observed_at":"2026-08-06T22:40:35.198292Z","submitted_at":"2026-05-08T18:40:47Z","title":"SecureForge: Finding and Preventing Vulnerabilities in LLM-Generated Code via Prompt Optimization","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-12T01:09:08.378040Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.08382"},"observation_digest":"sha256:86b29e57e50fd504200e93029041d925fe5aa770fcd116c2b88b950d314ae0c0","observation_id":"593fda4e-e88b-4287-a655-07e4b936771f","resolution":{"observed_at":"2026-05-12T08:26:24.588493Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.08898","last_updated":"2026-05-09T11:43:47Z","snapshot_observed_at":"2026-07-06T23:21:02.177557Z","submitted_at":"2026-05-09T11:43:47Z","title":"LLM-Agnostic Semantic Representation Attack","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-12T01:14:08.629862Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.08898"},"observation_digest":"sha256:8593b02b6947f8157747e3d3cb4c6b50f1bee6485cc1656f76182730b5eeb07c","observation_id":"d2708aee-a774-4683-95be-2aeb7caa45d7","resolution":{"observed_at":"2026-05-12T08:21:24.222423Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.17413","last_updated":"2026-05-17T12:18:20Z","snapshot_observed_at":"2026-08-03T01:59:45.581393Z","submitted_at":"2026-05-17T12:18:20Z","title":"Ablating Safety: Mechanisms for Removing Alignment in Language Models for Security Applications","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-19T23:30:43.364230Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.17413"},"observation_digest":"sha256:0d63dfa41b5c8d41a20a75d61127703cbb820a4d60e5412b772eada34bb5619b","observation_id":"de0b055b-e787-4943-bf8c-d0ea299bb904","resolution":{"observed_at":"2026-05-19T23:32:52.541609Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.20351","last_updated":"2026-05-19T18:05:51Z","snapshot_observed_at":"2026-07-06T23:30:58.549353Z","submitted_at":"2026-05-19T18:05:51Z","title":"Refusal Evaluation in Coding LLMs and Code Agents: A Systematic Review of Thirteen Malicious-Code Prompt Corpora (2023-2025)","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-21T07:38:43.363709Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.20351"},"observation_digest":"sha256:6d7be02a4b5eadd4ebe6f47f145d849666a2e015cdc288845792b8478e8a09e8","observation_id":"e9453341-ea23-434c-9bd0-857ce0aca0b2","resolution":{"observed_at":"2026-05-21T07:39:48.536733Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.21773","last_updated":"2026-05-20T22:07:12Z","snapshot_observed_at":"2026-07-06T23:32:10.472558Z","submitted_at":"2026-05-20T22:07:12Z","title":"HIDBench: Benchmarking Large Language Models for Host-Based Intrusion Detection","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T08:52:34.079804Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.21773"},"observation_digest":"sha256:60fe0a8726b6f9c9c4828b41666b76f0e6c790be2d5aa185200c76edab086b8c","observation_id":"014b4ae9-6471-42c6-9e79-be13ffc29e50","resolution":{"observed_at":"2026-05-22T08:54:45.868682Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T05:50:28.114140Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:db0d785ff88c4d8b869f9c52eff9c9d13dce4d0fa79c7146b441a13cd5e6f0ad","observation_id":"07a96601-9bbd-4e36-bdfd-14c9a7bae55b","resolution":{"observed_at":"2026-05-22T05:51:07.706475Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-07-06T23:32:59.663926Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-25T06:05:27.736494Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:53b4962420db1f5e9830e77505ff3f4e143b6fc4207ccf28db67b5621e953c2d","observation_id":"50142792-7fa7-41bc-bfa7-d3d3e189d660","resolution":{"observed_at":"2026-05-25T06:06:43.062405Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2605.23091","last_updated":"2026-05-21T22:53:40Z","snapshot_observed_at":"2026-08-01T20:58:37.779697Z","submitted_at":"2026-05-21T22:53:40Z","title":"Security of LLM-generated Code: A Comparative Analysis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T05:16:26.372764Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2605.23091"},"observation_digest":"sha256:2d13eec9baa45b49f0d35d1066f4a288b48e1566fc226e539cc7386bf73a1bba","observation_id":"8497cf12-ab6e-4c82-9ac5-566da15afb6a","resolution":{"observed_at":"2026-05-25T05:16:39.185029Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2606.01317","last_updated":"2026-05-31T16:06:02Z","snapshot_observed_at":"2026-08-10T05:39:13.866749Z","submitted_at":"2026-05-31T16:06:02Z","title":"SABER: Benchmarking Operational Safety of LLM Coding Agents in Stateful Project Workspaces","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-28T16:44:18.994680Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2606.01317"},"observation_digest":"sha256:d9a466a2faecab1e62419d6ba4059f4d4d85a99e6a9ca3ad0aa2734ca7504c66","observation_id":"0e47e741-ab83-4a5a-aa3c-4c83b18a6e4d","resolution":{"observed_at":"2026-07-01T21:36:15.066149Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2606.25973","last_updated":"2026-06-24T15:45:38Z","snapshot_observed_at":"2026-08-07T03:45:14.020128Z","submitted_at":"2026-06-24T15:45:38Z","title":"Helpful or Harmful? Evaluating LLM-Assisted Vulnerability Patching via a Human Study","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-25T19:02:45.109478Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2606.25973"},"observation_digest":"sha256:a2add03c41ef9ac47eed5df8867d935b32d4304226001bf286471240977a814d","observation_id":"e7a4f50f-7229-44a5-a12a-25664d9e796a","resolution":{"observed_at":"2026-07-04T21:10:09.288469Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":"2312.04724","doi":"10.48550/arxiv.2312.04724","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv:2312.04724 [cs]","venue":"arXiv (Cornell University)","work_id":"45b8079b-5204-450f-8024-f3a8142583a9","year":2023},"citing_paper":{"arxiv_id":"2606.29175","last_updated":"2026-06-28T03:43:30Z","snapshot_observed_at":"2026-07-07T00:03:12.330608Z","submitted_at":"2026-06-28T03:43:30Z","title":"Direct Causation in International Humanitarian Law and the Challenge of AI-Mediated Civilian Cyber Operations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T07:49:28.819402Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2606.29175"},"observation_digest":"sha256:c7a448e3d4d9e73cd9d404c208c66fbc6c02edfefebed3617c3b26bb68faafd2","observation_id":"9d048307-d19b-422f-a380-eb09b9fd25ee","resolution":{"observed_at":"2026-06-30T07:54:22.125211Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-07-12T07:33:43.015966Z","title":"PurpleLlama CyberSecEval: A secure coding benchmark for language models.arXiv preprint arXiv:2312.04724,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02714","last_updated":"2026-07-07T12:39:55Z","snapshot_observed_at":"2026-08-07T02:57:17.862073Z","submitted_at":"2026-07-02T19:05:07Z","title":"Not All Refusals Are Equal: How Safety Alignment Fails Cybersecurity at Scale","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-12T07:33:43.015966Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.02714"},"observation_digest":"sha256:4ae6f8437a0e9c0ccf9adfcebfd3da2c61e8201329bcea316cd133d203d5e8bc","observation_id":"f2caab62-2233-4b1d-a14f-600eca1074a7","resolution":{"observed_at":"2026-07-12T07:33:43.015966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-07-11T22:40:37.839133Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03968","last_updated":"2026-07-09T20:41:05Z","snapshot_observed_at":"2026-08-08T01:19:08.383401Z","submitted_at":"2026-07-04T17:57:05Z","title":"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-11T22:40:37.839133Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.03968"},"observation_digest":"sha256:f64495d052cc8ce3cdb7112586c2a26797b08bd08c7343e99a74b7899879273b","observation_id":"0d6179b5-8f42-4710-af49-7b22b97a1710","resolution":{"observed_at":"2026-07-11T22:40:37.839133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-07-13T07:01:49.222325Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03968","last_updated":"2026-07-09T20:41:05Z","snapshot_observed_at":"2026-08-08T01:19:08.383401Z","submitted_at":"2026-07-04T17:57:05Z","title":"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-13T07:01:49.222325Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.03968"},"observation_digest":"sha256:6da1e355aa59ac0615704ab21d906b5175d0e4d0819109b915db3f5a118e1f9a","observation_id":"c4b6b2a4-983a-46fb-a945-a3a224793c59","resolution":{"observed_at":"2026-07-13T07:01:49.222325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-01T02:40:29.099157Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25379","last_updated":"2026-08-01T10:03:47Z","snapshot_observed_at":"2026-08-06T23:11:27.985138Z","submitted_at":"2026-07-28T07:34:37Z","title":"Cyber-Capable AI Agents: Vulnerabilities, Evaluation Containment, and Defensive Response","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T02:40:29.099157Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.25379"},"observation_digest":"sha256:7331a5ca7b651c31ae1d6869cf757ac56751e2f56767ad1b6858e1c687b463e7","observation_id":"237e36ec-cebd-45cf-9c43-8a1d2b584881","resolution":{"observed_at":"2026-08-01T02:40:29.099157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-04T01:33:17.643282Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25379","last_updated":"2026-08-01T10:03:47Z","snapshot_observed_at":"2026-08-06T23:11:27.985138Z","submitted_at":"2026-07-28T07:34:37Z","title":"Cyber-Capable AI Agents: Vulnerabilities, Evaluation Containment, and Defensive Response","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T01:33:17.643282Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2607.25379"},"observation_digest":"sha256:1e1af3b0ed97792654e339426e5b7e2f9b24000926335c0de5bb57b11aa5445d","observation_id":"c2eb78e9-4e54-4e10-a6a0-07da0442523b","resolution":{"observed_at":"2026-08-04T01:33:17.643282Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.04724","snapshot_observed_at":"2026-08-08T19:58:17.193944Z","title":"Purple llama cyberseceval: A secure coding benchmark for language models.arXiv preprint arXiv:2312.04724, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04317","last_updated":"2026-08-05T00:54:57Z","snapshot_observed_at":"2026-08-09T19:45:40.565565Z","submitted_at":"2026-08-05T00:54:57Z","title":"Trident : How to Break Deep Reinforcement Learning Cyber Defenses (Agentic)","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T19:58:17.193944Z"},"links":{"cited_paper":"/paper/2312.04724","citing_paper":"/paper/2608.04317"},"observation_digest":"sha256:2323f7845cfe3d40dff70d0ce5813792cc06dd0e06488939637338a8e0c06f7b","observation_id":"7cdc2bd7-d8d6-4bb1-8996-da6ea68649a7","resolution":{"observed_at":"2026-08-08T19:58:17.193944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2312.04724/citation-record","integrity":"/paper/2312.04724/integrity","json":"/paper/2312.04724/citation-record.json","paper":"/paper/2312.04724"},"outbound":[],"paper":{"arxiv_id":"2312.04724","last_updated":"2023-12-07T22:07:54Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-09T23:28:16.743355Z","submitted_at":"2023-12-07T22:07:54Z","title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 44 inbound Pith citation observations for arXiv:2312.04724."}