{"as_of":"2026-08-15T23:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4cee0a9dc05fe9fbb8e9e7c3c7495a2d029d5981a1bb41da66094c6af59c09c6","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T23:19:22.732931Z","state":"measured"},{"denominator":34,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":34,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.16345/citation-record","integrity":"/paper/2607.16345/integrity","json":"/paper/2607.16345/citation-record.json","paper":"/paper/2607.16345"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.225374Z","title":"2025 , journal =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.225374Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:7eedb11ca1a6895aef09d2045e263f722ad45f9b22f5f91c599b9f6b2bc72bcf","observation_id":"21a85db5-e049-4b80-b5c4-9bbcff5b6fa6","resolution":{"observed_at":"2026-08-01T23:19:19.225374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.317735Z","title":"2005 , publisher =","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.317735Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:2a9fc8abdefbbaa698902132ff3ca0c69176b15316fbc2dc301ae557fbccf9bb","observation_id":"e00931a6-3420-4a0f-9123-0d979b8fb6a4","resolution":{"observed_at":"2026-08-01T23:19:19.317735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.459002Z","title":"Foundations and Trends in Machine Learning , volume =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.459002Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:bfe301b9b63861ff49622f4a4022fb8cbc870debbea8243e4ae3a87fa0270c1e","observation_id":"9b3fd911-9f53-4d5b-ba6c-1c1d6b664a53","resolution":{"observed_at":"2026-08-01T23:19:19.459002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.597770Z","title":"and Stoica, Ion and Zhang, Hao , booktitle =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.597770Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:3d093bcc46aa3cf9208cd51dcb4c9ecc9e40375d228f27c6ca513bb9661c1af4","observation_id":"334dbbda-5a0d-476f-a119-02935a02ec7c","resolution":{"observed_at":"2026-08-01T23:19:19.597770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.739760Z","title":"and Yang, John and Wettig, Alexander and Yao, Shunyu and Pei, Kexin and Press, Ofir and Narasimhan, Karthik , journal =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.739760Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:8df3b3d0f392f613e85f67242a58ed6e05976791ab1e21b57b1d48f20abba8ac","observation_id":"cd0ed951-7c64-4e56-9064-64f07af0ca49","resolution":{"observed_at":"2026-08-01T23:19:19.739760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.895023Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.895023Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:b98e4fbdd060eae6194e622e8f9f4edce5384b552ac4cf51f4029e2880af1cba","observation_id":"b57b168b-6a5b-4c2b-9848-d54c86678b23","resolution":{"observed_at":"2026-08-01T23:19:19.895023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:19.989747Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:19.989747Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:208fab255bf3d38e9d2e07541da5dfd1c425170fbfae0d3443dfa36b0870f33c","observation_id":"79ce5395-1d78-49c7-964e-2d67a4228a4b","resolution":{"observed_at":"2026-08-01T23:19:19.989747Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:20.078521Z","title":"2025 , howpublished =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:20.078521Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:ffedad99ef87a08002951053a1f513600e6e425b2cc9e62da2205faa0aef00eb","observation_id":"f10d89c6-83ed-4e71-9daf-a298a74cbd55","resolution":{"observed_at":"2026-08-01T23:19:20.078521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:20.222721Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:20.222721Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:bbc5fa18884c8db4625212c225487ed57e605931bb09c314bf2f0df933361fce","observation_id":"3f0d8440-7de3-47a9-9333-0ee5042716a9","resolution":{"observed_at":"2026-08-01T23:19:20.222721Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:20.398185Z","title":"2025 , howpublished =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:20.398185Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:2b2d016e807675df26fd0606d69d69d2b408e391d0db80af902f68481ae2341f","observation_id":"a69e7f7c-b49b-490a-84c0-bebc0d60727f","resolution":{"observed_at":"2026-08-01T23:19:20.398185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:20.509721Z","title":"2025 , howpublished =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:20.509721Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:60b7de242443d70194418c978bf86f97baaf2da6f94f7fcb759485e998633a2f","observation_id":"31b26334-5675-4a4c-be19-e7ee41df9547","resolution":{"observed_at":"2026-08-01T23:19:20.509721Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:20.931067Z","title":"Transactions on Machine Learning Research , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:20.931067Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:6551f680d93bd0bb0949f1ed13f664d9bf95403157d8380a8060ac61a5d514fe","observation_id":"1583d884-0fd5-4333-abb7-46c2baf4ec2c","resolution":{"observed_at":"2026-08-01T23:19:20.931067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.046645Z","title":"Science , volume =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.046645Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:59aca16a91ec9c05f167dd863f74f9f18857b86858520fd5630b21398a27be26","observation_id":"7ee97508-bec4-4bb6-9af8-27dc0a3157d0","resolution":{"observed_at":"2026-08-01T23:19:21.046645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.178463Z","title":"2025 , howpublished =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.178463Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:06cc67e7f20550be256a49ce753897510cc3f645c9b7315816022c1fc7350b9c","observation_id":"ef729ba6-6d37-4002-86f9-da62ebd6eea6","resolution":{"observed_at":"2026-08-01T23:19:21.178463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.237001Z","title":"and Feng, Shi , booktitle =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.237001Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:d32a21e600c3dae56409d4d8033a94b6a74dbc98ef0ca2e3a3c2684b48a7e00c","observation_id":"6ac764b5-e264-4f3f-af30-ec27a03677e7","resolution":{"observed_at":"2026-08-01T23:19:21.237001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.315062Z","title":"Constitutional","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.315062Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:4b7be2acc57cb389a92cd38ea992923d169d8fea9ce9daf17e4600097427527f","observation_id":"ef004a13-93be-4f33-a0c2-4f65dd41f47d","resolution":{"observed_at":"2026-08-01T23:19:21.315062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.375551Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.375551Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:5a7ae62d11f1cd6923093810b1dc401dbfb5fa9fd2efacd1f09e19236556fc69","observation_id":"1facdc1b-5180-4cbf-a6aa-609adde4cdcc","resolution":{"observed_at":"2026-08-01T23:19:21.375551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.445641Z","title":"N., Bates, S., Fannjiang, C., Jordan, M","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.445641Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:9cf938d3278d864891eac9a5631618a19ff96707a66ad6a6bacd77c32a03fefc","observation_id":"42a19c7c-6d91-457e-90d5-3da7d4dcf258","resolution":{"observed_at":"2026-08-01T23:19:21.445641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.517734Z","title":"Agent skills: Progressive disclosure for claude agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.517734Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:cddfe5e59af1d7bb24535770088f054114af2bd1ad7033e86c3a6098fba22c10","observation_id":"72eb76db-3beb-48a1-843f-c9e2589b8bc3","resolution":{"observed_at":"2026-08-01T23:19:21.517734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-01T23:19:21.599219Z","title":"Constitutional AI : Harmlessness from AI feedback","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.599219Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:0f163f803b2711701de29b9fb4008c98cf8d6c6ccbfb3dd743e948a996d2502d","observation_id":"b3f02336-3f59-4d83-92d5-4b442b64d2d7","resolution":{"observed_at":"2026-08-01T23:19:21.599219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.683068Z","title":"Braintrust: The enterprise-grade stack for building AI products","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.683068Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:5b301864b18c929d565fc606cf472b329fe1048c6c62ee3cea3c761564213990","observation_id":"8feb692e-b51a-461c-a6c1-e52d700ac562","resolution":{"observed_at":"2026-08-01T23:19:21.683068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:21.766482Z","title":"Harbor framework: Containerized evaluation for AI agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.766482Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:b3d9c970bba567d2237590b7a191373e9c2ac61c42f1c1b4607e6751edf16825","observation_id":"7d6fdc6f-a95d-4737-8d3a-1948c9eb49d2","resolution":{"observed_at":"2026-08-01T23:19:21.766482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-12T14:56:35.025839Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-01T23:19:21.880476Z","title":"E., Yang, J., Wettig, A., Yao, S., Pei, K., Press, O., and Narasimhan, K","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.880476Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:44bd011401fcd8f63d27f61095fad876608f924c83c28acd8cb6ca45b50fd744","observation_id":"b8ac7fc3-8677-4329-a7a9-77bd296f0e11","resolution":{"observed_at":"2026-08-01T23:19:21.880476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2207.05221","last_updated":"2022-11-21T16:38:35Z","snapshot_observed_at":"2026-08-14T05:42:29.067319Z","submitted_at":"2022-07-11T22:59:39Z","title":"Language Models (Mostly) Know What They Know","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.05221","snapshot_observed_at":"2026-08-01T23:19:21.943004Z","title":"Language models (mostly) know what they know","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:21.943004Z"},"links":{"cited_paper":"/paper/2207.05221","citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:d6d477239de6a03301a8d7d9e04609ec8af59912f3cba7e5b29382e33b0e2dcd","observation_id":"8d653346-829b-4e53-b780-ac675d3abee6","resolution":{"observed_at":"2026-08-01T23:19:21.943004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.024912Z","title":"LangSmith evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.024912Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:548a66d4998ad67688bc8238ba9e2b562c76dbf0fd6259201bc4f6db6c0d5469","observation_id":"cbc9eb41-b1fa-490d-b6f2-033cc1fa7e77","resolution":{"observed_at":"2026-08-01T23:19:22.024912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.108107Z","title":"Teaching models to express their uncertainty in words","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.108107Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:26537f89d66ba7835e5559cb256f0c5cb693e52ad819de9f791f91e27287bd64","observation_id":"41381c6f-29cd-4d83-a967-466b9708d950","resolution":{"observed_at":"2026-08-01T23:19:22.108107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.184407Z","title":"AgentBench : Evaluating LLM s as agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.184407Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:e1345b7da02bd0a39eee6a58a5bfaf524a65db6f0d88427106bef9948f8c76f7","observation_id":"0be6e847-34dd-4bbb-b09f-9a2399086cb3","resolution":{"observed_at":"2026-08-01T23:19:22.184407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.266063Z","title":"OpenAI evals: A framework for evaluating LLM s","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.266063Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:c876e2b77027ebe93887c2b7fe15c338304a5b2827f9ff956df4c9993884cce6","observation_id":"2cdcf771-5f19-44c6-bd81-5f20fa1f83a4","resolution":{"observed_at":"2026-08-01T23:19:22.266063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13076","last_updated":"2024-04-15T16:49:59Z","snapshot_observed_at":"2026-08-15T16:38:26.345282Z","submitted_at":"2024-04-15T16:49:59Z","title":"LLM Evaluators Recognize and Favor Their Own Generations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13076","snapshot_observed_at":"2026-08-01T23:19:22.347835Z","title":"R., and Feng, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.347835Z"},"links":{"cited_paper":"/paper/2404.13076","citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:1be8c0423449cd332e461e3e118e4a5c2c7c66d4a3c3b00e1f8b37fa61c2789d","observation_id":"94e32821-5dcc-4810-85e3-bcfd945ebb8f","resolution":{"observed_at":"2026-08-01T23:19:22.347835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.421080Z","title":"Promptfoo: Test your LLM app like software","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.421080Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:1e39a49e5e4adfe97eaf6053d05deb16b14ba21f0160ed880b8b0ea7a332a57c","observation_id":"6b479fbe-5176-4f17-adf4-bdd5ffcf2010","resolution":{"observed_at":"2026-08-01T23:19:22.421080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.09300","last_updated":"2023-12-14T19:09:22Z","snapshot_observed_at":"2026-08-14T20:38:43.961159Z","submitted_at":"2023-12-14T19:09:22Z","title":"Self-Evaluation Improves Selective Generation in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.09300","snapshot_observed_at":"2026-08-01T23:19:22.502759Z","title":"J., and Lakshminarayanan, B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.502759Z"},"links":{"cited_paper":"/paper/2312.09300","citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:ad072aeb068a0bfc4a2cb812be8e8817da3f2aa3a33f43a7a07f4176f2b2b778","observation_id":"14a11cff-b6e5-44eb-92c6-28e85ee47846","resolution":{"observed_at":"2026-08-01T23:19:22.502759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.05802","last_updated":"2022-06-14T01:16:24Z","snapshot_observed_at":"2026-08-13T17:24:30.719766Z","submitted_at":"2022-06-12T17:40:53Z","title":"Self-critiquing models for assisting human evaluators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.05802","snapshot_observed_at":"2026-08-01T23:19:22.583599Z","title":"Self-critiquing models for assisting human evaluators","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.583599Z"},"links":{"cited_paper":"/paper/2206.05802","citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:4b65b8c558e9294de52c1ba1c888431ea7e7ca334a58d6473ce333838cea6ee5","observation_id":"edd9e7ff-7691-411f-84c6-35870fcead45","resolution":{"observed_at":"2026-08-01T23:19:22.583599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.657597Z","title":"Algorithmic learning in a random world","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.657597Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:7beb09ec9841bbe72ef9d998650cbcb8c3ae65298e078869432ce862aa52ff9f","observation_id":"56c3a348-cb29-4f35-b802-ff00ae664a61","resolution":{"observed_at":"2026-08-01T23:19:22.657597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:19:22.732931Z","title":"E., Stoica, I., and Zhang, H","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-01T23:19:22.732931Z"},"links":{"citing_paper":"/paper/2607.16345"},"observation_digest":"sha256:cdb931b5fb462cb15526d170350f8b33d201e026426996e33877544e0778d4a0","observation_id":"48079d4b-3710-4a50-9f24-8df7a26f7ba6","resolution":{"observed_at":"2026-08-01T23:19:22.732931Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.16345","last_updated":"2026-07-21T00:27:08Z","latest_version":2,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-06T02:36:47.417860Z","submitted_at":"2026-07-16T21:33:05Z","title":"AEVAL: From Anecdotal to Deterministic Testing for Agentic Skill Workflows"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":2,"unresolved":32,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 0 inbound Pith citation observations for arXiv:2607.16345."}