{"as_of":"2026-08-10T14:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f87e2463a8cc0a757d2ab5904acd3c56744d7c439256449fd2ef4b805fc95b2b","coverage":[{"denominator":87,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":87,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T10:20:42.139206Z","state":"measured"},{"denominator":87,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":87,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.05573/citation-record","integrity":"/paper/2608.05573/integrity","json":"/paper/2608.05573/citation-record.json","paper":"/paper/2608.05573"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.840481Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.840481Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:04f63d8b47893482a8eb5d465f62443c33bf8cc7d35263f24fee3287092b152d","observation_id":"e1838b19-9ae6-4b53-9b74-849d764d42a1","resolution":{"observed_at":"2026-08-08T10:20:41.840481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.844288Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.844288Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:3daec4e432a8148038591155124e68259d9de4fb78d4c2b57552debc78094cbf","observation_id":"d6746e21-fe57-4773-99be-8f88edde0e95","resolution":{"observed_at":"2026-08-08T10:20:41.844288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.847861Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.847861Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:97fcf653fa5005da78ae4e9058539d3e0d916ae0d09a5509ffd4c880b63af467","observation_id":"2dc189fb-9cbd-464f-827b-10e9c99b6d06","resolution":{"observed_at":"2026-08-08T10:20:41.847861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.851210Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.851210Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:5757e7bb2a59171e7d25c09a7de591830c9326c70746131224ef15f98a811533","observation_id":"115d3e34-87b1-4a40-8cff-2ca1193eacae","resolution":{"observed_at":"2026-08-08T10:20:41.851210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.854672Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.854672Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:640e43d989fc6a51de989e4cc95d933abc1a1541b995217c6511f2ae65c212c8","observation_id":"36bc2a8c-a90a-44de-8ec9-87d43b5345fe","resolution":{"observed_at":"2026-08-08T10:20:41.854672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.858096Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.858096Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:5ffcf6b2ff15b3759ded66608beac625db6958ddb2544f63a62fd8f36e06f802","observation_id":"605723e7-9dd6-4b43-8317-2d4fd92ac338","resolution":{"observed_at":"2026-08-08T10:20:41.858096Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.861913Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.861913Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:3734d2ab2b661393da30604f9a498d46d65dde28e3d4092e70d618b13c9c5dfa","observation_id":"76a16a61-8e05-4305-80c2-a9854e4b31c8","resolution":{"observed_at":"2026-08-08T10:20:41.861913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.865188Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.865188Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:d9b305c990bb384fabeb3c6d4085da65bd03712b76848745868e3ff3ea7b58d0","observation_id":"4124fb18-2c70-4e82-aaf4-b0c893cd52f8","resolution":{"observed_at":"2026-08-08T10:20:41.865188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.868568Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.868568Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:ac8c5966f178c95c0357814337d8dcd0ddf136c8d654fb658d4c483c69014ca3","observation_id":"c758cc47-ffcb-44dd-8a6a-0a3985549a56","resolution":{"observed_at":"2026-08-08T10:20:41.868568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.871676Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.871676Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:9f971fdd4160cf109b2fe41edd4de42653c9ad45efc00b633fde96be6f66532c","observation_id":"2f265538-767b-4323-b80f-6de9cbc71207","resolution":{"observed_at":"2026-08-08T10:20:41.871676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.884039Z","title":"2023 , eprint=","venue":null,"work_id":"16a9485a-bfbe-46e4-a805-ae1d814e0bec","year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.874949Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:377757aa3a83b64ed4afa88af6cf98bf76b9594146868f00695e3b2b844dbf47","observation_id":"d8f3d3ea-f70d-43b8-ad0d-60f8f493f376","resolution":{"observed_at":"2026-08-08T10:20:43.886907Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.878151Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.878151Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:bed1936202106baa71fa2f5f5392baf4e6a4bcd38a1a678fe19320f764b7f0ee","observation_id":"fa0e07a9-5873-4a23-b1e9-4a0a38016745","resolution":{"observed_at":"2026-08-08T10:20:41.878151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.881504Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.881504Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:e088b3d822f0388dad4b0e7a73601ade4b375bce99ed2f9f14221d54ad3c1d75","observation_id":"c2d5d464-9c85-4661-b3a1-ce9ff103b120","resolution":{"observed_at":"2026-08-08T10:20:41.881504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.885487Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.885487Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:dea3a7249ae0548d7abfe0365fbc18398d0a8495bd7b6aa0cf6e7e08e43b78fc","observation_id":"901690f8-2cfb-44f1-bfb8-63d6244c504a","resolution":{"observed_at":"2026-08-08T10:20:41.885487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.888670Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.888670Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:dd9cd20af6a731987e1b4a50bc592675e37401c2f87a6f371969516bfe544693","observation_id":"71f7a4e3-e1fa-433a-8013-b1c38591f2d6","resolution":{"observed_at":"2026-08-08T10:20:41.888670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.852364Z","title":"American Journal of Physics , volume=","venue":null,"work_id":"fe445fa3-e43f-4659-9de2-5d68f2257202","year":null},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.891952Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:8aad0a5fd9dfcf018da92c2e0a3ac527911c3a596db1ee9de714d99bd9666c0e","observation_id":"1dd45944-d872-4523-a882-0b80e3f12839","resolution":{"observed_at":"2026-08-08T10:20:43.855529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.895829Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.895829Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:1250c76b4bc5740a394acbec5e21501379f0790e9949f7872f4cf7a73806e6f2","observation_id":"16f3e26a-adec-4f14-a051-53f9ee6a090e","resolution":{"observed_at":"2026-08-08T10:20:41.895829Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.836623Z","title":"2026 , eprint=","venue":null,"work_id":"8c7b6f31-8ca9-4c5e-895c-699f9feb62de","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.899389Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:db94aebe258b22548d13d88741aa67c30ae8c970fc503187df4725041b3018dc","observation_id":"f786dde9-5592-4bab-b3db-ae816f438117","resolution":{"observed_at":"2026-08-08T10:20:43.840466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.902619Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.902619Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:1dec145172d007c2f6e50324692ddfc65486a267d159f317e054c574e45b7a8d","observation_id":"9a75888d-d688-494c-a431-133612daaca6","resolution":{"observed_at":"2026-08-08T10:20:41.902619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.905914Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.905914Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:420cc36519053dc266f9053d4b8daaf4d4f13bd3f80f190ad84d52a7361cb29b","observation_id":"23fb6f24-093d-4c11-b0ae-cf3f32436750","resolution":{"observed_at":"2026-08-08T10:20:41.905914Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.909605Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.909605Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:03cf0df4737f23c038b4d2470fad59b746726258b314cf88b22fd7af5b4c0338","observation_id":"20915c10-3a66-441e-a6ad-d8862cb4c515","resolution":{"observed_at":"2026-08-08T10:20:41.909605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.912952Z","title":"2025 , eprint=","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.912952Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:8861add133276e0c7f5512e8c9e5b175293ee720da4817f87906687bcebfc554","observation_id":"73bba7e4-7e6a-4eac-a93b-536537643724","resolution":{"observed_at":"2026-08-08T10:20:41.912952Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.804449Z","title":"2025 , eprint=","venue":null,"work_id":"503dcdee-f683-4763-8f26-ce3d6c4a8517","year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.916102Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:d7dfc76ddec81797d58ef2cc5f35ffcfd4e08d2a6bcd8ab8e618d9c0d891552c","observation_id":"1814d7fd-ca8b-4c96-8dcd-0d9378370ca3","resolution":{"observed_at":"2026-08-08T10:20:43.807975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.794344Z","title":"2026 , eprint=","venue":null,"work_id":"26502865-813d-43c8-b083-aba63ee47190","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.919673Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:baadb8b75f49d35d5b675602e114e5b1ba20d1a38f7b75a889eea878a9cdbdc2","observation_id":"aaff6c34-0d1b-4008-a4ef-01bca91a3c62","resolution":{"observed_at":"2026-08-08T10:20:43.797956Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.784029Z","title":"2025 , eprint=","venue":null,"work_id":"4282632b-ecb5-4fe6-b608-d078fa530791","year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.923029Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:46104e65e94887da522d3f6574977a41bcdd8f05f5d94c080d50359bef673b3b","observation_id":"4311a4f0-41f2-438b-928f-dc2c00302e75","resolution":{"observed_at":"2026-08-08T10:20:43.787533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.773994Z","title":"2026 , eprint=","venue":null,"work_id":"c7c33401-0934-4831-9c21-85ba5062bb18","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.926170Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:a6d8d855767ea5f4cfcd098a845d4a02bfdf48186269eaa740a48a027e5f16c5","observation_id":"5d0f2b5c-1b21-472e-a54d-261d131047d2","resolution":{"observed_at":"2026-08-08T10:20:43.777775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.764184Z","title":"2026 , eprint=","venue":null,"work_id":"0cbbfa5f-aaf3-4359-9a83-7b84f6f3e7b2","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.929460Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:4d988fd52296c1aac39f7fe1b95855cb17edcba26d9f635dc9753d3e71d15565","observation_id":"46133c43-a6ae-4a56-a2c5-d36a7f450d01","resolution":{"observed_at":"2026-08-08T10:20:43.767756Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.754063Z","title":"2026 , eprint=","venue":null,"work_id":"ef2febb9-b619-41a6-a698-1cedeefaa348","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.932800Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:35b6917eedad1c7c8baac67cc704c97172a3a67b7b580dbb9c40e8ada0fad7ac","observation_id":"46934b6c-ab7b-481b-9c2e-9d93de6f1191","resolution":{"observed_at":"2026-08-08T10:20:43.757626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.743259Z","title":"2025 , eprint=","venue":null,"work_id":"fdd896c5-2ba4-4b1e-901f-c7f42971ce32","year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.936365Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:937f2400318d4e4c200f6dcf5b10465d7a6272e2c6b198a41c78b5f286985e2f","observation_id":"45cb4063-7c6f-4b5d-98c0-8e598c14ceed","resolution":{"observed_at":"2026-08-08T10:20:43.746771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.939386Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.939386Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:be17d46f6ed4c53c2aa42fd6495e06326c08f7597a15795ecdc4b8639e1686c6","observation_id":"2f72aca4-d14b-40ae-8a6d-d0c71cbc56bd","resolution":{"observed_at":"2026-08-08T10:20:41.939386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.942919Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.942919Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:e2025671907bea9bee9982d5bda590706c0d0417f6a8d952c614d3eb3654179e","observation_id":"5f1d7a2f-0ba1-4b4b-b3ba-30cf16fedd09","resolution":{"observed_at":"2026-08-08T10:20:41.942919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.721952Z","title":null,"venue":null,"work_id":"a707c8ee-2e03-49a6-a8f0-b705b930c3e4","year":null},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.949776Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:978da757b00873a0ded0d6a8969a854ca25e986ffb116e7d54f10f510d436da2","observation_id":"3ccb06a9-53be-4771-9d5b-c493ad7efec8","resolution":{"observed_at":"2026-08-08T10:20:43.725545Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.952912Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.952912Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:d063f1b0277ed657512fca314b0ece3abead518bd48b8db90d8b07e3f5a19691","observation_id":"0c2fca53-b17b-41f0-9807-9185cd688b15","resolution":{"observed_at":"2026-08-08T10:20:41.952912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.955776Z","title":"2025 , month = dec, howpublished =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.955776Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:eadc25f208e6cdb020b6ca77787f93bbee1095b2a1c8eb36a7ac59efc436edf4","observation_id":"c7ce7d7d-5917-4404-827b-eef02ff86665","resolution":{"observed_at":"2026-08-08T10:20:41.955776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.958642Z","title":"2026 , month = feb, howpublished =","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.958642Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:fc5c7a36251ea9b4a9c85c075f4c70bf47f600ddb103e1b746e14f6544e8b52a","observation_id":"3137ec12-ed27-466f-bda3-ec7535043a49","resolution":{"observed_at":"2026-08-08T10:20:41.958642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.961697Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.961697Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:8f47693ab55764491c148176b0edd552a273309b86f67ab44a244e89e7da7d20","observation_id":"9e023d98-eae1-4613-905d-3cd269a0cb94","resolution":{"observed_at":"2026-08-08T10:20:41.961697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.964844Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.964844Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:019380c6852170032dd1714abbd2ee51633d441a004090cd78ed3fcefa3be3a9","observation_id":"aded129b-ba23-485b-ab89-0f491432c2c7","resolution":{"observed_at":"2026-08-08T10:20:41.964844Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.684925Z","title":"2026 , eprint=","venue":null,"work_id":"c0d9a35a-e576-4b1a-8efd-9754f6b84665","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.967762Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:7cc762ca4b264d738377566cfc2a2bb551f546ce4bd54aa0b1ba3a3b62ac093c","observation_id":"ef923420-b94c-4741-9cb1-c5c198fa28f9","resolution":{"observed_at":"2026-08-08T10:20:43.688386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.674914Z","title":"2026 , eprint=","venue":null,"work_id":"de3a59b5-1085-4608-8aeb-099b9f647934","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.971179Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:475cf4b2ddbcd085b4b6655016287be342e2042bfe010065e287853d168a80f6","observation_id":"5c57f621-1b2f-4cd8-b98c-25ca48ea0b68","resolution":{"observed_at":"2026-08-08T10:20:43.678640Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.665658Z","title":"2026 , eprint=","venue":null,"work_id":"b2b289fc-27b4-4391-af98-3a06654d182f","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.974473Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:b1edf911477f9972e67754ef59825f23eb18dcd70208bef8e59adff678e2cbac","observation_id":"31d8c9d7-45c7-4681-9f31-d376959d28a3","resolution":{"observed_at":"2026-08-08T10:20:43.668642Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.656861Z","title":"2026 , eprint=","venue":null,"work_id":"78156b3e-4d2d-41dd-8d73-5a7334b7fd02","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.977614Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:64adf173d2e588ca6407cab89c7f22faeedf799b6697df727f9b507acf27c211","observation_id":"6c6fbaed-d074-42f9-b59a-6a94f937b1d1","resolution":{"observed_at":"2026-08-08T10:20:43.659779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.981047Z","title":"2024 , eprint=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.981047Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:ac97ac78dc8f028baf8534bf5e37e2c71ea1111cedc55959469e18282a88d61d","observation_id":"f1e1222e-e8f7-4d92-873e-4db3afc80033","resolution":{"observed_at":"2026-08-08T10:20:41.981047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.984603Z","title":"2023 , eprint=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.984603Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:f8ac76b6bc1fe827c95ecf5bbe977a838813167a99486c8d5e9fb60d39ef95b6","observation_id":"a77bf0bc-47f4-4456-ab0a-70735985353c","resolution":{"observed_at":"2026-08-08T10:20:41.984603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.634311Z","title":"2025 , eprint=","venue":null,"work_id":"ffd53b5a-9c53-4442-9e25-4445d73041a4","year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.988036Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:377ede7540eac8cb7330d711e4f4ebaefd5f7930af42447760d7a43558301521","observation_id":"60064129-dc20-4a03-ad04-bc0f3c5d9b3b","resolution":{"observed_at":"2026-08-08T10:20:43.638229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.622225Z","title":"2026 , eprint=","venue":null,"work_id":"63f2c5e2-4489-486a-8740-85f8c74eecdb","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.991087Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:478e84fe85e358bfae7bff7b5c44e5f5ae79427a9559d628232391bc21140f56","observation_id":"4467e8d7-58dd-454f-935c-f9a7b9504359","resolution":{"observed_at":"2026-08-08T10:20:43.626085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:41.994447Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.994447Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:15ea0a844822e2ac1972435603a80411f17b73900c0a02660437a7a71c54773d","observation_id":"63e28695-0423-4b24-be9c-bb376043ed32","resolution":{"observed_at":"2026-08-08T10:20:41.994447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19457","last_updated":"2026-02-14T11:42:30Z","snapshot_observed_at":"2026-08-10T14:31:17.392776Z","submitted_at":"2025-07-25T17:42:32Z","title":"GEPA: Reflective Prompt Evolution Can Outperform Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.19457","snapshot_observed_at":"2026-08-08T10:20:41.997995Z","title":"A.; Tan, S.; Soylu, D.; Ziems, N.; Khare, R.; Opsahl-Ong, K.; Singhvi, A.; Shandilya, H.; Ryan, M","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:41.997995Z"},"links":{"cited_paper":"/paper/2507.19457","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:73de99c1d60edcf2fa62e266669cbd3822016c744f1ec0405bf26af033106aab","observation_id":"00163140-ff00-42e1-993b-f26dc19b6bfd","resolution":{"observed_at":"2026-08-08T10:20:41.997995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.604042Z","title":null,"venue":null,"work_id":"4ca96e69-de69-4ef8-bfe5-77b10121db3a","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.001840Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:e15d552482bfc03a674cd0e6483e866dcae621a2dc7e5682955f3b32cfaefbce","observation_id":"0db83985-a646-47f7-94c3-c790b4d5cb5c","resolution":{"observed_at":"2026-08-08T10:20:43.607735Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.005062Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.005062Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:7f34add0100cbfc4b083814514bb749861d1d0f6d60956cfadd06913fb99ff87","observation_id":"03b43aa9-ea6a-472b-9b34-296cd4262595","resolution":{"observed_at":"2026-08-08T10:20:42.005062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21787","last_updated":"2024-12-30T19:03:24Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:57:25Z","title":"Large Language Monkeys: Scaling Inference Compute with Repeated Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21787","snapshot_observed_at":"2026-08-08T10:20:42.008420Z","title":"V.; Ré, C.; and Mirhoseini, A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.008420Z"},"links":{"cited_paper":"/paper/2407.21787","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:2e9fbe959ee9ce10486405326ef711e01b9c016c0ed10deda2b90f49199d64db","observation_id":"a939bd00-543f-4d0a-9997-4734aa44cc00","resolution":{"observed_at":"2026-08-08T10:20:42.008420Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.03980","last_updated":"2026-06-02T17:56:57Z","snapshot_observed_at":"2026-07-06T23:44:09.692359Z","submitted_at":"2026-06-02T17:56:57Z","title":"Skill-RM: Unifying Heterogeneous Evaluation Criteria via Agent Skill","version":1},"cited_work":{"arxiv_id":"2606.03980","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.03980","snapshot_observed_at":"2026-08-08T10:20:43.383865Z","title":"Skill-RM: Unifying Heterogeneous Evaluation Criteria via Agent Skill","venue":"cs.LG","work_id":"dec0622e-2f88-46f9-8d01-937d6c56ec09","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.012024Z"},"links":{"cited_paper":"/paper/2606.03980","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:702b0bccafa021f3ebdfb8e397eaf37db2917c284303791f2f822bee43b74d96","observation_id":"7e53c890-7176-4a94-be53-a6486cfcfd1d","resolution":{"observed_at":"2026-08-08T10:20:43.390067Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.11543","last_updated":"2026-06-10T01:11:50Z","snapshot_observed_at":"2026-08-08T14:39:34.388821Z","submitted_at":"2026-06-10T01:11:50Z","title":"SkillJuror: Measuring How Agent Skill Organization Changes Runtime Behavior","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.11543","snapshot_observed_at":"2026-08-08T10:20:42.015720Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.015720Z"},"links":{"cited_paper":"/paper/2606.11543","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:983c8f3f72fce73d51a709c9288759e5389328da1b67ae38a94e570f714c0519","observation_id":"99ce2704-f7f9-4a00-b166-14c7106acc03","resolution":{"observed_at":"2026-08-08T10:20:42.015720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08638","last_updated":"2025-06-23T21:06:11Z","snapshot_observed_at":"2026-08-07T15:45:43.526005Z","submitted_at":"2025-05-13T14:55:31Z","title":"TRAIL: Trace Reasoning and Agentic Issue Localization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.08638","snapshot_observed_at":"2026-08-08T10:20:42.019625Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.019625Z"},"links":{"cited_paper":"/paper/2505.08638","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:18e2c7191aab43ddc84df23e9e913743c099a4a4481fff1a37dcb43ee2aa0c84","observation_id":"a3e682bc-6a56-477b-8050-229d5f06caae","resolution":{"observed_at":"2026-08-08T10:20:42.019625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.11435","last_updated":"2026-06-09T20:43:23Z","snapshot_observed_at":"2026-08-06T07:30:17.162850Z","submitted_at":"2026-06-09T20:43:23Z","title":"Agent Skill Evaluation and Evolution: Frameworks and Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.11435","snapshot_observed_at":"2026-08-08T10:20:42.023057Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.023057Z"},"links":{"cited_paper":"/paper/2606.11435","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:68c0043880d4e9e7ac616dbb80e6663572cb53e0c1edb2de4c3a8db20be8d86a","observation_id":"77931dce-e424-4f4b-ba1b-cb2ba15ac6ab","resolution":{"observed_at":"2026-08-08T10:20:42.023057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07718","last_updated":"2024-07-23T06:19:28Z","snapshot_observed_at":"2026-08-02T12:09:24.340284Z","submitted_at":"2024-03-12T14:58:45Z","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07718","snapshot_observed_at":"2026-08-08T10:20:42.027123Z","title":"H.; Verme, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.027123Z"},"links":{"cited_paper":"/paper/2403.07718","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:74560f53d7e77aecfb0672290b07258170ec6bf9ed3c8681843863bdeecc26a0","observation_id":"f43a4304-974a-4340-a9b2-42e524b03d4b","resolution":{"observed_at":"2026-08-08T10:20:42.027123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16797","last_updated":"2023-09-28T19:01:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T19:01:07Z","title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16797","snapshot_observed_at":"2026-08-08T10:20:42.030214Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.030214Z"},"links":{"cited_paper":"/paper/2309.16797","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:6759f8728c9f1c0c01f1314c8f3f42172376aa6b877b1e90bf6ed1d352f0419f","observation_id":"4027aed4-c4ec-422f-bf7b-b3e8db0287ca","resolution":{"observed_at":"2026-08-08T10:20:42.030214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.033779Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.033779Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:48c47cac10366b9f00f323604f0f83a61abff636d7ee863637d05c8fd33f6508","observation_id":"2b2b70a5-17a1-470a-b6ff-61047b3f4d51","resolution":{"observed_at":"2026-08-08T10:20:42.033779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.036730Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.036730Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:1085fc83aff4515cc59c4c42a92e9d27ee28c91cc163b6539423f056da303388","observation_id":"509a09f3-8f4e-42c6-ba7c-3f437d975017","resolution":{"observed_at":"2026-08-08T10:20:42.036730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01535","last_updated":"2024-12-04T19:23:17Z","snapshot_observed_at":"2026-08-06T22:44:08.607902Z","submitted_at":"2024-05-02T17:59:35Z","title":"Prometheus 2: An Open Source Language Model Specialized in Evaluating Other Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01535","snapshot_observed_at":"2026-08-08T10:20:42.039474Z","title":"Y.; Shin, J.; Welleck, S.; Neubig, G.; Lee, M.; Lee, K.; and Seo, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.039474Z"},"links":{"cited_paper":"/paper/2405.01535","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:b2a010d60c262385370af5423017923ac8e554bbe51e10b3ef39e7a65b8c91c6","observation_id":"6be1e988-f0e5-4429-bf44-975420dc87b3","resolution":{"observed_at":"2026-08-08T10:20:42.039474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13787","last_updated":"2024-06-08T16:40:12Z","snapshot_observed_at":"2026-08-02T18:11:57.036767Z","submitted_at":"2024-03-20T17:49:54Z","title":"RewardBench: Evaluating Reward Models for Language Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13787","snapshot_observed_at":"2026-08-08T10:20:42.042569Z","title":"Y.; Chandu, K.; Dziri, N.; Kumar, S.; Zick, T.; Choi, Y.; Smith, N","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.042569Z"},"links":{"cited_paper":"/paper/2403.13787","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:12cfdb3dd7894dd72cd3ee7687b2ffc56657159fb2f593452692a8ca9f15a5e5","observation_id":"a65af639-b90b-491a-b3b5-a2ed11e28685","resolution":{"observed_at":"2026-08-08T10:20:42.042569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12670","last_updated":"2026-03-13T07:33:01Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T07:06:06Z","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.12670","snapshot_observed_at":"2026-08-08T10:20:42.046260Z","title":"W.; Sun, J.; Wang, S.; Tao, C.; Li, B.; Zhao, X.; Geng, H.; Wu, X.; Zhou, J.; Chen, X.; Xing, H.; Li, Y.; Zeng, Q.; Wang, D.; Wang, Y.; Chaim, R","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.046260Z"},"links":{"cited_paper":"/paper/2602.12670","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:63492e2297730e4d11d96283959baa6c5dd141f887099f18a2acfede601e674d","observation_id":"df2e98db-e610-41f0-8d71-f28ee9b8ca9f","resolution":{"observed_at":"2026-08-08T10:20:42.046260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.049441Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.049441Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:305e3173ff294949de2a9dd97b188ca5bf64e4cc60073847cb972030cff01840","observation_id":"dc2babe8-a99b-403d-9c29-63d0d47ffee3","resolution":{"observed_at":"2026-08-08T10:20:42.049441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16634","last_updated":"2023-05-23T22:12:16Z","snapshot_observed_at":"2026-08-02T04:02:36.848064Z","submitted_at":"2023-03-29T12:46:54Z","title":"G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.16634","snapshot_observed_at":"2026-08-08T10:20:42.052412Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.052412Z"},"links":{"cited_paper":"/paper/2303.16634","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:e8ec92c6e1bcf1d74e9a219b8909355d466a93b21e76cef815029fddef522f92","observation_id":"57f586e4-c838-4d27-aa1c-84a528e5cf7c","resolution":{"observed_at":"2026-08-08T10:20:42.052412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.01139","last_updated":"2026-06-17T05:10:07Z","snapshot_observed_at":"2026-07-06T23:41:43.820102Z","submitted_at":"2026-05-31T10:19:13Z","title":"SkillRevise: Improving LLM-Authored Agent Skills via Trace-Conditioned Skill Revision","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.01139","snapshot_observed_at":"2026-08-08T10:20:42.055705Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.055705Z"},"links":{"cited_paper":"/paper/2606.01139","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:d2b980c1b43e80c642293ac93ceb5ca3ef52c1895afc4f154658eeb32efe37da","observation_id":"62a3a2b6-d3a9-4617-95f5-b5edce64e164","resolution":{"observed_at":"2026-08-08T10:20:42.055705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.13143","last_updated":"2025-08-18T17:55:22Z","snapshot_observed_at":"2026-08-05T19:06:25.616496Z","submitted_at":"2025-08-18T17:55:22Z","title":"Exploring Autonomous Agents: A Closer Look at Why They Fail When Completing Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.13143","snapshot_observed_at":"2026-08-08T10:20:42.058799Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.058799Z"},"links":{"cited_paper":"/paper/2508.13143","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:09e431b315cca36cbdf6c98b41cb8dcfab54402389f362ce5f843d0c2521f142","observation_id":"4a4d8dca-9e04-46be-bc1c-95eef9990c49","resolution":{"observed_at":"2026-08-08T10:20:42.058799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.061941Z","title":"H.; Kazemnejad, A.; Meade, N.; Patel, A.; Shin, D.; Zambrano, A.; Stańczak, K.; Shaw, P.; Pal, C","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.061941Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:0a6f4f6318a19484a0de3c4844eadeceddd03b3983610b7baec2b4c8b6227474","observation_id":"a7399c01-5fca-49af-a7de-a568a0a36584","resolution":{"observed_at":"2026-08-08T10:20:42.061941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.064892Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.064892Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:a3efaa7a8103e867e23298ee0e7aea1aa124939dbb93c4e9f3cde491ea382527","observation_id":"76fe7547-9f91-4b51-be64-4965d5138f0a","resolution":{"observed_at":"2026-08-08T10:20:42.064892Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-08T10:20:42.068370Z","title":"A.; Shaw, A","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.068370Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:07dfc133c6cb6a68ab65ab7b09c7ba7be2395a3ce3fd672ab9cfe65e4328b7d3","observation_id":"d902ca15-77bb-4c25-ba82-86a530a72e0a","resolution":{"observed_at":"2026-08-08T10:20:42.068370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.592099Z","title":null,"venue":null,"work_id":"f2170e99-3d69-47fb-9483-5f76f18ea7b1","year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.072021Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:6b8931c3aaccc8e5fc0564e46de6569e83fac74b75415e9a364864901750c374","observation_id":"a0be9b6c-63d0-4895-8b98-8901043a00ac","resolution":{"observed_at":"2026-08-08T10:20:43.595893Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.075461Z","title":"W.; Liu, J.; Chen, W.; Chen, Z.; and Lou, Y","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.075461Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:91c914cbdaf22a7b4f57d85e894db39f58b6032b817145d549fb486b20e6bd74","observation_id":"414a0229-1d5c-4ef0-9ceb-e187ecd9b577","resolution":{"observed_at":"2026-08-08T10:20:42.075461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.04761","last_updated":"2023-02-09T16:49:57Z","snapshot_observed_at":"2026-07-06T14:50:07.491434Z","submitted_at":"2023-02-09T16:49:57Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.04761","snapshot_observed_at":"2026-08-08T10:20:42.078991Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.078991Z"},"links":{"cited_paper":"/paper/2302.04761","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:3afa00645a2236409d247c09fb1b6881e48b99fd6eb8236af25ad938a91e80ed","observation_id":"75b3d18f-7b98-46be-b496-57586c529a14","resolution":{"observed_at":"2026-08-08T10:20:42.078991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.06847","last_updated":"2026-05-07T16:46:46Z","snapshot_observed_at":"2026-08-08T08:47:19.105155Z","submitted_at":"2026-03-06T20:12:29Z","title":"Characterizing Faults in Agentic AI: A Taxonomy of Types, Symptoms, and Root Causes","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.06847","snapshot_observed_at":"2026-08-08T10:20:42.082598Z","title":"B.; Morovati, M","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.082598Z"},"links":{"cited_paper":"/paper/2603.06847","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:0419a0a911ba01061a2f08750580aa8bc8387fffa2a5894fe00f7d67bd5667be","observation_id":"1260d50e-e60a-4d04-aff8-03836dfcd4f3","resolution":{"observed_at":"2026-08-08T10:20:42.082598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18240","last_updated":"2026-04-20T13:23:38Z","snapshot_observed_at":"2026-07-06T23:05:08.728446Z","submitted_at":"2026-04-20T13:23:38Z","title":"AJ-Bench: Benchmarking Agent-as-a-Judge for Environment-Aware Evaluation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.18240","snapshot_observed_at":"2026-08-08T10:20:42.086849Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.086849Z"},"links":{"cited_paper":"/paper/2604.18240","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:a82b261b1bef2d159ddc0502708ac7e47285eb2f05c4722c7dba6b365bc1ea35","observation_id":"212e6fc4-a23a-4a68-83d9-b0dd0d1f55d8","resolution":{"observed_at":"2026-08-08T10:20:42.086849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03314","last_updated":"2024-08-06T17:35:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:35:05Z","title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03314","snapshot_observed_at":"2026-08-08T10:20:42.090825Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.090825Z"},"links":{"cited_paper":"/paper/2408.03314","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:af1211976f06e6324152879f9ba730ab4cc2805b9a1d6be3639d05a2a7736afa","observation_id":"0e83ceb7-74c6-45c0-ab31-30937fb559cd","resolution":{"observed_at":"2026-08-08T10:20:42.090825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.19504","last_updated":"2025-08-27T01:29:46Z","snapshot_observed_at":"2026-08-05T15:41:43.211333Z","submitted_at":"2025-08-27T01:29:46Z","title":"Aegis: Taxonomy and Optimizations for Overcoming Agent-Environment Failures in LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.19504","snapshot_observed_at":"2026-08-08T10:20:42.094789Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.094789Z"},"links":{"cited_paper":"/paper/2508.19504","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:a59714272b3d5ca993e3afae6a5746db0023dc5d7dc76a252de170d0878627ec","observation_id":"fe62e75e-1728-43a8-81e8-54901ceac076","resolution":{"observed_at":"2026-08-08T10:20:42.094789Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12784","last_updated":"2025-04-05T00:07:35Z","snapshot_observed_at":"2026-07-31T04:18:41.491400Z","submitted_at":"2024-10-16T17:58:19Z","title":"JudgeBench: A Benchmark for Evaluating LLM-based Judges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12784","snapshot_observed_at":"2026-08-08T10:20:42.098785Z","title":"Y.; Cuadron, A.; Wang, C.; Popa, R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.098785Z"},"links":{"cited_paper":"/paper/2410.12784","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:ccfc8720212cac4cc864d506c6441da734a7bc46bfa770311cd4c57ebda18099","observation_id":"81177f5d-6c6b-4a08-bdec-73c47b27ea40","resolution":{"observed_at":"2026-08-08T10:20:42.098785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:43.580678Z","title":null,"venue":null,"work_id":"87c6b750-018f-4c3f-ae5a-ed093217c560","year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.102691Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:56b3f2815028f9fc57c177fcf428d645e8cb0d0fef3c4c897f40007ebce1b6ea","observation_id":"b3b1ab29-11f5-49eb-89f4-f191c1ddb722","resolution":{"observed_at":"2026-08-08T10:20:43.584544Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.15808","last_updated":"2026-04-29T08:40:46Z","snapshot_observed_at":"2026-07-06T22:42:43.581560Z","submitted_at":"2026-01-22T09:47:31Z","title":"Inference-Time Scaling of Verification: Self-Evolving Deep Research Agents via Test-Time Rubric-Guided Verification","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.15808","snapshot_observed_at":"2026-08-08T10:20:42.106403Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.106403Z"},"links":{"cited_paper":"/paper/2601.15808","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:0448d8c11273d5093b7838c7917b3df05dc0db070afb7e7d84ee604766c70127","observation_id":"4998afc2-02ec-415c-8f4c-b9609f1d8b84","resolution":{"observed_at":"2026-08-08T10:20:42.106403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08178","last_updated":"2026-05-11T07:26:22Z","snapshot_observed_at":"2026-08-02T16:02:49.097271Z","submitted_at":"2026-04-09T12:35:06Z","title":"Aligning Agents via Planning: A Benchmark for Trajectory-Level Reward Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.08178","snapshot_observed_at":"2026-08-08T10:20:42.111062Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.111062Z"},"links":{"cited_paper":"/paper/2604.08178","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:a44a44b54f46bdcb3b5d2c062e2e420617021e543fcf662d175b56a51adce1ad","observation_id":"b635b442-6671-4c56-87de-87b47a2168b1","resolution":{"observed_at":"2026-08-08T10:20:42.111062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.03409","last_updated":"2024-04-15T07:50:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-07T00:07:15Z","title":"Large Language Models as Optimizers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.03409","snapshot_observed_at":"2026-08-08T10:20:42.114847Z","title":"V.; Zhou, D.; and Chen, X","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.114847Z"},"links":{"cited_paper":"/paper/2309.03409","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:a36a8dfbe68516372d0f8f82aaa93966c15460fef234286715fffe0e45d3d87f","observation_id":"49cebb2b-7740-416a-a37f-6136e0684a27","resolution":{"observed_at":"2026-08-08T10:20:42.114847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-08T21:08:39.676079Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-08T10:20:42.118832Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.118832Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:62654b0cbe27957225c004a2b1d4fab494b7d91548f2d1d162f2fc555209749f","observation_id":"0423ab39-9a53-47c7-8675-5b716629344d","resolution":{"observed_at":"2026-08-08T10:20:42.118832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-08T10:20:42.122793Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.122793Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:7443065fde850076a3b5b15bb565b7660fe1dd06f24ee3840f31139a91693f76","observation_id":"f645ac63-7a16-4883-bd6f-6f19ced31253","resolution":{"observed_at":"2026-08-08T10:20:42.122793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.20087","last_updated":"2026-04-22T01:07:37Z","snapshot_observed_at":"2026-07-06T23:06:35.507008Z","submitted_at":"2026-04-22T01:07:37Z","title":"SkillLearnBench: Benchmarking Continual Learning Methods for Agent Skill Generation on Real-World Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.20087","snapshot_observed_at":"2026-08-08T10:20:42.126671Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.126671Z"},"links":{"cited_paper":"/paper/2604.20087","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:f9c91cbe3aa72e00b3ef3214ccba6342fab961d9eb0a8b4724c63f923aa7659e","observation_id":"673bbefc-2f79-44a7-8952-0da07ea5c559","resolution":{"observed_at":"2026-08-08T10:20:42.126671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12928","last_updated":"2025-06-15T17:59:47Z","snapshot_observed_at":"2026-08-09T00:02:34.658764Z","submitted_at":"2025-06-15T17:59:47Z","title":"Scaling Test-time Compute for LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.12928","snapshot_observed_at":"2026-08-08T10:20:42.130163Z","title":"E.; Zhang, C.; Lin, C.; Wang, J.; Zhang, G.; and Zhou, W","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.130163Z"},"links":{"cited_paper":"/paper/2506.12928","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:7c54d0e1ea62627ab83bf7f0981c15fccd1ba872ea7a7ce6a22c7d45ebfc6e01","observation_id":"3e914543-90f7-4f96-bbdc-5af9c64d2bb7","resolution":{"observed_at":"2026-08-08T10:20:42.130163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T10:20:42.132968Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.132968Z"},"links":{"citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:b7c22ab06af0ea69df1898f203429a49690b90e3dd63216dc393cd3df383e942","observation_id":"ba1d707d-f51f-4b5c-83a7-a5c9c0e1bbba","resolution":{"observed_at":"2026-08-08T10:20:42.132968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.21806","last_updated":"2026-08-06T07:48:13Z","snapshot_observed_at":"2026-08-09T23:09:31.931445Z","submitted_at":"2026-02-25T11:34:17Z","title":"Where Agent Frameworks Fall Short: Examining Functional Challenges and Usability Concerns","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.21806","snapshot_observed_at":"2026-08-08T10:20:42.136007Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.136007Z"},"links":{"cited_paper":"/paper/2602.21806","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:ffef34aede4cfa8f5b235164d4c80fd9f385aa2c7e04757f0138b8e089de2bdb","observation_id":"84a52ccf-4e28-4d6c-8625-14447c1bc5e2","resolution":{"observed_at":"2026-08-08T10:20:42.136007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10934","last_updated":"2024-10-16T17:54:12Z","snapshot_observed_at":"2026-08-08T15:14:50.065788Z","submitted_at":"2024-10-14T17:57:02Z","title":"Agent-as-a-Judge: Evaluate Agents with Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10934","snapshot_observed_at":"2026-08-08T10:20:42.139206Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-08T10:20:42.139206Z"},"links":{"cited_paper":"/paper/2410.10934","citing_paper":"/paper/2608.05573"},"observation_digest":"sha256:1412bcdadff4d54c755f4732296e691ec5b65d40189db9a2b5c7864f3a4400b4","observation_id":"04bacc21-0956-4a45-b058-15d960ea72f2","resolution":{"observed_at":"2026-08-08T10:20:42.139206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.05573","last_updated":"2026-08-06T03:52:37Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T23:11:05.903937Z","submitted_at":"2026-08-06T03:52:37Z","title":"SkillTV-Bench: Benchmarking How Well Judges Perform on Skill-Augmented Agentic Execution"},"reference_resolution":{"displayed":87,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":70,"verified_exact":1,"verified_fuzzy":16},"total_outbound_references":87},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 87 of 87 outbound references and 0 inbound Pith citation observations for arXiv:2608.05573."}