{"as_of":"2026-08-21T02:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:80d11498443a1f0fa1f717e29bd80823aec0c7c256a0ffc15a05a1d9badcd20f","coverage":[{"denominator":20,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":20,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:19:09.555981Z","state":"measured"},{"denominator":20,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":20,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.11469/citation-record","integrity":"/paper/2608.11469/integrity","json":"/paper/2608.11469/citation-record.json","paper":"/paper/2608.11469"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.18624","last_updated":"2024-07-10T05:26:17Z","snapshot_observed_at":"2026-08-16T14:05:52.069753Z","submitted_at":"2024-03-27T14:34:29Z","title":"Vulnerability Detection with Code Language Models: How Far Are We?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18624","snapshot_observed_at":"2026-08-15T14:19:09.463443Z","title":"Vulnerability detection with code language models: How far are we?arXiv preprint arXiv:2403.18624,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.463443Z"},"links":{"cited_paper":"/paper/2403.18624","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:da57f657e83248882f29a4763e4aff4e95274c58a968c0a1ad2c3b4b8283671e","observation_id":"b1eb8ad7-90b9-4218-b1a5-c45a73908c59","resolution":{"observed_at":"2026-08-15T14:19:09.463443Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:10.111996Z","title":"Livecodebench: Holistic and contamination free evalua- tion of large language models for code","venue":null,"work_id":"1793148b-e28c-4594-b26a-ee788bef4e7f","year":2025},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.482962Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:f089f14dd0d496a86742ab145551c348e22a31b380761e2f6e5459da7fc1e65b","observation_id":"b91c5c13-c24f-4c0f-b0f9-44e36b867e0b","resolution":{"observed_at":"2026-08-15T14:19:10.117159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.05493","last_updated":"2026-06-03T22:34:15Z","snapshot_observed_at":"2026-08-13T04:22:28.641538Z","submitted_at":"2026-06-03T22:34:15Z","title":"REStack: A Large-Scale Dataset of Reverse Engineering Discussions from Stack Exchange","version":1},"cited_work":{"arxiv_id":"2606.05493","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.05493","snapshot_observed_at":"2026-08-15T14:19:09.989255Z","title":"REStack: A Large-Scale Dataset of Reverse Engineering Discussions from Stack Exchange","venue":"cs.SE","work_id":"74b74eb6-d31b-4fa7-a945-eda3b28cd6eb","year":2026},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.488937Z"},"links":{"cited_paper":"/paper/2606.05493","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:f47eb20b6a7f2d8e8b6c43c432b4b9cf612f17c7aac6fa1ed958172739625151","observation_id":"add1ed68-6e4b-4d14-a6cd-5e7bcb1d6cca","resolution":{"observed_at":"2026-08-15T14:19:09.995528Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2607.07738","last_updated":"2026-07-24T13:29:41Z","snapshot_observed_at":"2026-08-16T23:34:43.691234Z","submitted_at":"2026-07-07T23:17:03Z","title":"REFORGE: A Method for Benchmarking LLMs' Reverse Engineering Capabilities in Decompiled Binary Function Naming","version":2},"cited_work":{"arxiv_id":"2607.07738","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.07738","snapshot_observed_at":"2026-08-15T14:19:09.959877Z","title":"REFORGE: A Method for Benchmarking LLMs' Reverse Engineering Capabilities in Decompiled Binary Function Naming","venue":"cs.SE","work_id":"a39f717b-8cf6-4e0a-9b40-50c9ecf3e0d1","year":2026},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.494438Z"},"links":{"cited_paper":"/paper/2607.07738","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:a94f8b6ab3ac2f7f645b17644011573d91ba2c7a64e860d84bc31ad9d795e575","observation_id":"89e9e2b8-9263-4ed2-ba8a-f2fbeccc797a","resolution":{"observed_at":"2026-08-15T14:19:09.966622Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.26548","last_updated":"2026-07-20T16:24:12Z","snapshot_observed_at":"2026-08-17T17:28:45.182891Z","submitted_at":"2026-05-26T04:59:49Z","title":"SEC-bench Pro: Can Language Models Solve Long-Horizon Software Security Tasks?","version":2},"cited_work":{"arxiv_id":"2605.26548","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.26548","snapshot_observed_at":"2026-08-15T14:19:09.925136Z","title":"SEC-bench Pro: Can Language Models Solve Long-Horizon Software Security Tasks?","venue":"cs.CR","work_id":"9ca4f556-7662-4950-9347-7e260e0a1ace","year":2026},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.500035Z"},"links":{"cited_paper":"/paper/2605.26548","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:53a44d6f590a0cf61d2ba50e0c99ec9eaef1448eea186385c6728f6e2231e63c","observation_id":"fd51eebc-b6e7-4fdf-b5c8-a32a1b2c1bfc","resolution":{"observed_at":"2026-08-15T14:19:09.932406Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.14153","last_updated":"2026-05-13T22:08:05Z","snapshot_observed_at":"2026-08-16T08:24:32.940937Z","submitted_at":"2026-05-13T22:08:05Z","title":"ExploitBench: A Capability Ladder Benchmark for LLM Cybersecurity Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.14153","snapshot_observed_at":"2026-08-15T14:19:09.506467Z","title":"Exploitbench: A capability ladder benchmark for llm cyberse- curity agents.arXiv preprint arXiv:2605.14153,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.506467Z"},"links":{"cited_paper":"/paper/2605.14153","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:acc2e882b0433392b4f697c04d5de60463f303b0b61dc47e86f8f2621e547f2e","observation_id":"c552c635-48c1-4632-b541-da7b525265cf","resolution":{"observed_at":"2026-08-15T14:19:09.506467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07595","last_updated":"2024-08-21T14:51:06Z","snapshot_observed_at":"2026-08-16T13:44:10.631840Z","submitted_at":"2024-06-11T13:42:57Z","title":"VulDetectBench: Evaluating the Deep Capability of Vulnerability Detection with Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07595","snapshot_observed_at":"2026-08-15T14:19:09.511903Z","title":"Vulde- tectbench: Evaluating the deep capability of vulnerability detection with large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.511903Z"},"links":{"cited_paper":"/paper/2406.07595","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:384c4709c0f767adc0eb1666c651eee1d90a94ed70eab096fb9c9c633d901a81","observation_id":"3dafe5fe-d4a7-48a8-a715-060c674c64e5","resolution":{"observed_at":"2026-08-15T14:19:09.511903Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:09.517108Z","title":"Patch-to-poc: A systematic study of agentic llm systems for linux kernel n-day reproduction.arXiv preprint arXiv:2602.07287,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.517108Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:00d360f4ba8c8dc512cf8278560d0f8902a11e01cc526c538264107a38ca4883","observation_id":"2385ced4-2f48-4612-9641-74581e10eab3","resolution":{"observed_at":"2026-08-15T14:19:09.517108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.11086","last_updated":"2026-05-11T18:00:14Z","snapshot_observed_at":"2026-08-16T23:33:20.820875Z","submitted_at":"2026-05-11T18:00:14Z","title":"ExploitGym: Can AI Agents Turn Security Vulnerabilities into Real Attacks?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.11086","snapshot_observed_at":"2026-08-15T14:19:09.533769Z","title":"Exploitgym: Can ai agents turn security vulnerabilities into real attacks?arXiv preprint arXiv:2605.11086,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.533769Z"},"links":{"cited_paper":"/paper/2605.11086","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:47880a663eb2ff2934220024b17c33d7187dbfb57e46951bfe78199124d16ccc","observation_id":"33f44ef8-e248-460e-82d0-050a37b5e4c2","resolution":{"observed_at":"2026-08-15T14:19:09.533769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.27319","last_updated":"2026-04-30T02:03:37Z","snapshot_observed_at":"2026-08-14T21:26:47.572039Z","submitted_at":"2026-04-30T02:03:37Z","title":"REBENCH: A Procedural, Fair-by-Construction Benchmark for LLMs on Stripped-Binary Types and Names (Extended Version)","version":1},"cited_work":{"arxiv_id":"2604.27319","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.27319","snapshot_observed_at":"2026-08-15T14:19:09.710688Z","title":"REBENCH: A Procedural, Fair-by-Construction Benchmark for LLMs on Stripped-Binary Types and Names (Extended Version)","venue":"cs.CR","work_id":"0319f818-b2f4-4608-a704-693139932a16","year":2026},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.538836Z"},"links":{"cited_paper":"/paper/2604.27319","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:d4db653d9cca8c2850a91d4c5725ae75395c033c230a94e00d2642d996cc735e","observation_id":"59ea09b2-ab52-4a31-920b-e7bd3b2cc4d7","resolution":{"observed_at":"2026-08-15T14:19:09.718401Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:10.078592Z","title":"Cybench: A framework for evaluating cyber- security capabilities and risks of language models","venue":null,"work_id":"36c40582-c7b7-405d-86b2-05789c955ef5","year":2025},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.544423Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:53ea253c6d01fcb9412bdb536c4859327aa420162e5cb36aabfb12685c3e4a17","observation_id":"fbfa9057-47ef-4d9a-9aec-a9d5f361fe27","resolution":{"observed_at":"2026-08-15T14:19:10.084545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17332","last_updated":"2025-06-24T04:10:59Z","snapshot_observed_at":"2026-08-20T05:27:00.273588Z","submitted_at":"2025-03-21T17:32:32Z","title":"CVE-Bench: A Benchmark for AI Agents' Ability to Exploit Real-World Web Application Vulnerabilities","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.17332","snapshot_observed_at":"2026-08-15T14:19:09.549877Z","title":"Cve-bench: a benchmark for ai agents’ ability to exploit real-world web application vulnerabilities.arXiv preprint arXiv:2503.17332,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.549877Z"},"links":{"cited_paper":"/paper/2503.17332","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:9e0904d51ab552a7ec02b87177eb6804620c86a14a161e32d95e92dea957ce19","observation_id":"868f2da6-28b3-4df1-bd39-6bb51e0ea382","resolution":{"observed_at":"2026-08-15T14:19:09.549877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:09.555981Z","title":"Training language model agents to find vulnerabilities with ctf-dojo.arXiv preprint arXiv:2508.18370,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.555981Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:7417a3351e578c4334ec38da3b242fe77442982b1b932fc8147faea64b54b82d","observation_id":"735dac0f-0924-4810-95f8-6c070051ae83","resolution":{"observed_at":"2026-08-15T14:19:09.555981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.03750","last_updated":"2026-08-06T02:00:49Z","snapshot_observed_at":"2026-08-14T19:43:00.848314Z","submitted_at":"2026-04-04T14:51:09Z","title":"CREBench: Evaluating Large Language Models in Cryptographic Binary Reverse Engineering","version":2},"cited_work":{"arxiv_id":"2604.03750","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.03750","snapshot_observed_at":"2026-08-15T14:19:10.054680Z","title":"CREBench: Evaluating Large Language Models in Cryptographic Binary Reverse Engineering","venue":"cs.CR","work_id":"3abf37e7-d696-4478-99a5-965ee9bb9a60","year":2026},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":1994,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.451651Z"},"links":{"cited_paper":"/paper/2604.03750","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:cedc57fafca28badd7dbf7b980b9c1ede75e60116ac354407d5ac23eb55faf75","observation_id":"1aa92109-06f1-4f88-88dd-7ba8b870c5d6","resolution":{"observed_at":"2026-08-15T14:19:10.063822Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:10.169922Z","title":"The concept assignment problem in program understanding","venue":null,"work_id":"74a9d991-b93c-4556-80d6-44aff3fbc9b5","year":1993},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":2005,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.441883Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:f2371a69190b8c77af8579c204e68690951d6f7af285b7e9a6d1211477d2394e","observation_id":"1efe0ece-f99c-458b-bf16-0dbf85826279","resolution":{"observed_at":"2026-08-15T14:19:10.175322Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:10.095941Z","title":"Benchmarking binary type inference techniques in decompilers","venue":null,"work_id":"77f3433b-b6e6-457d-81c4-ddd448c49fdd","year":2025},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":2008,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.521963Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:0ae584c59681ad0ef35462710b654baaf323002766db6b3ace76e6bf90254008","observation_id":"c75fbb3c-ea5c-49f6-8c25-ae639e94bf46","resolution":{"observed_at":"2026-08-15T14:19:10.101141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:10.146745Z","title":"De- compilebench: A comprehensive benchmark for evaluating decompilers in real-world scenarios","venue":null,"work_id":"7527717e-5f19-4d44-ad88-853ef5a9ecd7","year":2025},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":2011,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.470783Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:5c6de44fe186be3ecb27f178efbf4e63bc18973df9e2c796fc3e63cfb90fa071","observation_id":"2292012d-bf99-4722-addb-1cc959017dc0","resolution":{"observed_at":"2026-08-15T14:19:10.154010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:09.528096Z","title":"Cy- bergym: Evaluating ai agents’ real-world cybersecurity capabilities at scale.arXiv preprint arXiv:2506.02548,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.528096Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:968f66d9020362c80debdadb3929e6ca8bb2e42b5b02397490e5b478793d5431","observation_id":"34d9a0fe-8c3f-4e3e-9226-5a02dc55cb1d","resolution":{"observed_at":"2026-08-15T14:19:09.528096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T14:19:10.128758Z","title":"Look what you made us patch: 2025 zero-days in review.https://cloud.google","venue":null,"work_id":"7abf90de-a091-464d-8d01-ebd736f04830","year":2025},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.476129Z"},"links":{"citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:e1cf99dd2b7f3b13b088405668e93abf9034fde7789258f30a588ce1e27803c5","observation_id":"8574ff04-1778-4338-95ee-1823f150f486","resolution":{"observed_at":"2026-08-15T14:19:10.134531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.10597","last_updated":"2026-05-11T14:01:36Z","snapshot_observed_at":"2026-08-11T13:11:49.987032Z","submitted_at":"2026-05-11T14:01:36Z","title":"CrackMeBench: Binary Reverse Engineering for Agents","version":1},"cited_work":{"arxiv_id":"2605.10597","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.10597","snapshot_observed_at":"2026-08-15T14:19:10.031250Z","title":"CrackMeBench: Binary Reverse Engineering for Agents","venue":"cs.SE","work_id":"390195e3-8323-43d3-99b2-5d5f37be3e00","year":2026},"citing_paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:09.458239Z"},"links":{"cited_paper":"/paper/2605.10597","citing_paper":"/paper/2608.11469"},"observation_digest":"sha256:2ee38112272d3895515122b7bb6455324d27d1089575490a8a8140443abe7f60","observation_id":"919065b6-03ce-4b70-968e-669227f06a61","resolution":{"observed_at":"2026-08-15T14:19:10.035977Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.11469","last_updated":"2026-08-11T22:14:57Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-18T22:13:32.845752Z","submitted_at":"2026-08-11T22:14:57Z","title":"The Next Challenge for Agentic Cybersecurity: A Realistic, Contamination-Free Reverse Engineering Benchmark"},"reference_resolution":{"displayed":20,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":8,"verified_exact":6,"verified_fuzzy":6},"total_outbound_references":20},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 20 of 20 outbound references and 0 inbound Pith citation observations for arXiv:2608.11469."}