{"as_of":"2026-08-20T18:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7353dc9b4d390d085102c7b31003abf9bb8afdbaf27ec16b762c67ea1ebb7e88","coverage":[{"denominator":62,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":62,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-10T00:02:58.383840Z","state":"measured"},{"denominator":62,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":62,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.06873/citation-record","integrity":"/paper/2607.06873/integrity","json":"/paper/2607.06873/citation-record.json","paper":"/paper/2607.06873"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.387063Z","title":"ReAct: Synergizing reasoning and acting in language models","venue":null,"work_id":"dcede5c1-a91b-43d9-a097-8083603cb625","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:ebe16b281725f942faa2e525107f1e668b90e53b5d02aa41d79b10cec6fc8ada","observation_id":"8c9d1e7c-c348-422f-946b-62e9568cbc46","resolution":{"observed_at":"2026-07-10T00:06:38.388221Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.388780Z","title":"Toolformer: Language models can teach themselves to use tools,","venue":null,"work_id":"d55164da-4150-4f21-99ae-664e11d9652a","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:b970055bf418b998d4df0fb989d964f411975759488c42dd8f0f9b14d174c621","observation_id":"72f10df0-056f-4523-8e0e-6aca88842e4b","resolution":{"observed_at":"2026-07-10T00:06:38.389906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-17T20:31:29.818313Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:18c60aa9ee501667715f592525fb80462e1a378183c5b04a7458e10164494b04","observation_id":"4ea94b3d-f9b9-4888-9852-dd8a529e6290","resolution":{"observed_at":"2026-07-10T00:06:37.954013Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.392206Z","title":"Preventing repeated real world AI failures by cataloging incidents: The AI incident database,","venue":null,"work_id":"04f82ceb-1e89-4947-ab4f-907b61439719","year":2021},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c701e7745abe45fc4e6a209fe7450f3cc0638eb827a075854cb8127918242cce","observation_id":"4a4dfd04-19c2-4c43-a2fe-eafb3ce7c78b","resolution":{"observed_at":"2026-07-10T00:06:38.393408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.351995Z","title":"RealHarm: A collection of real-world language model application failures,","venue":null,"work_id":"680a6106-96d1-47eb-8a5a-845608a9fe22","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:671c43a2fab203b4f82e8ba68dbafc59a7da0a71654acd45c36e251e2bfdbef2","observation_id":"4ffa99de-f735-4887-b6f9-e719cb61aae6","resolution":{"observed_at":"2026-07-10T00:06:38.353168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-08-14T06:34:01.114459Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":"2506.07982","doi":"10.48550/arxiv.2506.07982","metadata_source":"pith","pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","venue":"cs.AI","work_id":"3a498b1a-455f-4667-b572-c5216c99a89c","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:186e415337ca417d4deb8884627e68e4614ee3cb7fc4117bd77fba437f9ac5eb","observation_id":"f27b15c7-3df5-4d10-bf94-f07e62d6107d","resolution":{"observed_at":"2026-07-10T00:06:37.972449Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.383510Z","title":"WebArena: A realistic web environment for building autonomous agents,","venue":null,"work_id":"4496153c-9436-47b2-8cca-6c1643d00a18","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3a13ed49a2f0b12a69e9784ac4c9899ca842ff02d24ca9248f531e662ca66390","observation_id":"e592fa41-3f52-480f-a16a-9c01898fae04","resolution":{"observed_at":"2026-07-10T00:06:38.384707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.380080Z","title":"GAIA: A benchmark for general AI assistants,","venue":null,"work_id":"1f850405-8593-4d88-98f8-8f90ef2150ef","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:2a353638cc5c857c0b0ab565719eafbe7ae90f9b314f358b23974220a6048755","observation_id":"0f3b2126-34aa-4a65-826f-ade350a7d323","resolution":{"observed_at":"2026-07-10T00:06:38.381201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.376539Z","title":"AppWorld: A controllable world of apps and people for benchmarking interactive coding agents,","venue":null,"work_id":"8401e33f-476e-48e4-839d-a27859811428","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3218fc6cf530bb9a14b132a0c515dcf9b6a0c25b4dd6cb6a8ac6d6a300582920","observation_id":"99a03855-90bb-4410-81dd-38744782f243","resolution":{"observed_at":"2026-07-10T00:06:38.377808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.399395Z","title":"SWE-bench: Can language models resolve real-world GitHub issues?","venue":null,"work_id":"67f71a17-451e-4c9d-8866-589d5e7e2244","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3af4626867dd0b4d373e96a4e956fe6de6f3d5148c0c9c849051e099444785a5","observation_id":"c9221d46-79f0-4923-8502-f38c9021ec18","resolution":{"observed_at":"2026-07-10T00:06:38.401263Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.397731Z","title":null,"venue":null,"work_id":"b0c27d99-18e0-48f6-a698-8ed6c098b86f","year":2011},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c827db396d442b2d27ef00a532efea9abc8aae06e83b527e9d7a3937189b81ce","observation_id":"1f0725ff-4b9e-447f-b1c5-f27a6c1c8728","resolution":{"observed_at":"2026-07-10T00:06:38.398826Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.346356Z","title":"SpecOps: A fully automated AI agent testing framework in real-world GUI environments,","venue":null,"work_id":"85d8a534-c0b7-4823-b150-63893fe2f821","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:b2f223d2ef03825eb593d03a24167e5dafa69e36beba372d6d6660412ecda8c5","observation_id":"63c1db74-4e01-44c4-8c8c-8f522b8203c8","resolution":{"observed_at":"2026-07-10T00:06:38.347539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.00497","doi":"10.48550/arxiv.2601.00497","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"STELLAR: A search- based testing framework for large language model applications","venue":"arXiv (Cornell University)","work_id":"21d233de-26f8-492a-9344-6c57697c5195","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3536cc569e962bd8fc041bc9fe62c069a4b7b348e38138c3011e497b36e60724","observation_id":"daea19d1-e896-4fbd-9314-21ce8a47e832","resolution":{"observed_at":"2026-07-10T00:06:37.967164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.419584Z","title":null,"venue":null,"work_id":"9cc11dfe-752e-46e4-a37b-d3872a4f2e60","year":2016},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c57fd225586490595450ce336f74017e4b9d71388668e7de8d80a1bb5e75b17c","observation_id":"6143456a-537e-4313-8166-7a7f273e6ff7","resolution":{"observed_at":"2026-07-10T00:06:38.420670Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.421338Z","title":"A practitioner’s guide to process mining: Limitations of the directly-follows graph,","venue":null,"work_id":"2d8bfcd8-d517-433e-a59e-add1bdc570ef","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:427ac49d098334816f4ba684d6d356157ea71b477d403acffe08b012740776c4","observation_id":"aec1170e-470f-4be9-8296-7486691ec429","resolution":{"observed_at":"2026-07-10T00:06:38.422481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.428153Z","title":"τ 3-bench: From text-only to multimodal, knowledge- aware agent evaluation,","venue":null,"work_id":"4ac8eb2f-cec1-42d8-b2cf-516f215d4046","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:0f567d5ccf0a16428bcfde42ff1378314fd5ad465b25e3b950dedaa788930dbe","observation_id":"57d39246-2796-4bff-b036-68890eadf922","resolution":{"observed_at":"2026-07-10T00:06:38.429430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.422960Z","title":"Event abstraction for process mining using supervised learning techniques,","venue":null,"work_id":"1414aa58-ab58-46bd-919e-d7df6f406848","year":2016},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:4d930c0160e39a55880a56d90486d0c97e653c82f1525de95fc6ec3f34a383de","observation_id":"e5e93c59-a7ef-4463-9592-d07d6a3f5697","resolution":{"observed_at":"2026-07-10T00:06:38.424048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.407607Z","title":"Event abstraction in process mining: Literature review and taxonomy,","venue":null,"work_id":"f5994e0a-62ea-4130-b1d0-64f4646a900a","year":2021},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:11ff2b86029f329a84e6d3f2959fcc11829ea2d7631a05d415212a0e992eac85","observation_id":"94bedfb8-7f23-49ae-8e03-092a82ad01ac","resolution":{"observed_at":"2026-07-10T00:06:38.408812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.412680Z","title":"Boundary value exploration for software analysis,","venue":null,"work_id":"737aa211-77aa-4e22-9b71-43d5d488df13","year":2020},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:72c0d794807d24728d36b8447b41cdf8120e7e5fd102b81403f693e9313bb04e","observation_id":"a3e665c4-f53a-48ac-830b-15ce92ec3fb1","resolution":{"observed_at":"2026-07-10T00:06:38.413741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.426467Z","title":"Automated robustness testing of off-the-shelf software components,","venue":null,"work_id":"e8bc83a9-4dc1-4a81-86e1-389d20e5375f","year":1998},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:ea9f3e6563ec7f700ed23ed11a45bef48bd9b0fe9863b2c43397c3c716b4ebd8","observation_id":"94441f16-9357-421f-bc16-0c2ed032bfb8","resolution":{"observed_at":"2026-07-10T00:06:38.427615Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.04370","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.974347Z","title":"τ- knowledge: Evaluating conversational agents over unstructured knowl- edge,","venue":null,"work_id":"0b88f354-50d7-46c9-bb81-2cce36a131b9","year":2026},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:be7ce8030ccd157134ee04c6285ad991923f544f086d49c1e4ef6844c794fc29","observation_id":"954f9e5c-be91-4ff4-8818-3fd3c6f4b45d","resolution":{"observed_at":"2026-07-10T00:06:37.976247Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.390182Z","title":"ToolLLM: Facilitating large language models to master 16000+ real-world APIs,","venue":null,"work_id":"2d6e4ae2-a6b3-420b-9bb2-03aad7226356","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:4d1818213ba155a555437f7f0178e94698dd3b92e89ca5e569f2e41d67f2911a","observation_id":"528726a0-4bbc-43f0-82da-02e1076c96fc","resolution":{"observed_at":"2026-07-10T00:06:38.391348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.00415","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.961746Z","title":"Towards self-evolving benchmarks: Synthesizing agent trajectories via test-time exploration under validate-by-reproduce paradigm","venue":null,"work_id":"447ad45f-cd0f-4901-8512-0c35542bde4a","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:7477c69c653b4b4ec8c7474fb410bfaabf496051595b110dc89b41b57af01eca","observation_id":"313d6d59-d163-401c-9181-18e859c0f448","resolution":{"observed_at":"2026-07-10T00:06:37.963929Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2410.11507","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.973618Z","title":"Revisiting benchmark and assessment: An agent-based exploratory dynamic evaluation framework for llms","venue":null,"work_id":"25f98d0b-1a74-494f-b6be-a9120415b2b8","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a1879dbed5a56ca452f01d1a663e626a66ef57190bef1e13a8dc816ada7fd327","observation_id":"97062318-6ecb-476d-8c13-da94c9bc37b9","resolution":{"observed_at":"2026-07-10T00:06:37.975111Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.00507","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:37.971374Z","title":"Graph2Eval: Automatic multimodal task generation for agents via knowledge graphs,","venue":null,"work_id":"6d3ef2d5-4034-405b-bcdd-f205d16da8a9","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:446b1cca3b2638e07fdb5d2d7950441983c8886a5de44bd30b35813c201eec6a","observation_id":"86574250-54b4-41bf-a838-365082b4f56f","resolution":{"observed_at":"2026-07-10T00:06:37.973251Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17968","last_updated":"2025-05-23T14:37:36Z","snapshot_observed_at":"2026-08-14T14:54:33.659117Z","submitted_at":"2025-05-23T14:37:36Z","title":"Are Large Language Models Reliable AI Scientists? Assessing Reverse-Engineering of Black-Box Systems","version":1},"cited_work":{"arxiv_id":"2505.17968","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17968","snapshot_observed_at":"2026-07-10T00:06:37.955246Z","title":"arXiv preprint arXiv:2505.17968 , year=","venue":"cs.LG","work_id":"0d9a4447-37ad-49f8-99b3-19c44722d6c8","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2505.17968","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:bdee6481ec2b5a599ff63ee40bd707cb7a136bb863c68d4ab74e62c4561dd27e","observation_id":"5bddc95b-cf75-4d8b-bd2c-3374077b922a","resolution":{"observed_at":"2026-07-10T00:06:37.956842Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.371344Z","title":"Mining specifications,","venue":null,"work_id":"94c18af8-ec44-4c2c-b965-4f9e37831437","year":2002},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:fcb183db491d447561cb412a3792411618a032cc12b5d98cfeac9ce32ba41b1d","observation_id":"bdd38177-ce3c-4b83-9b07-36928981b4da","resolution":{"observed_at":"2026-07-10T00:06:38.372426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.393917Z","title":"Discovering models of software processes from event-based data,","venue":null,"work_id":"27132c23-9870-4008-854a-590d85298ea6","year":1998},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:8c57244706c404e77d5ef7ba0195bfbfbae2be34463e8b1032094448cdba478c","observation_id":"84bb6543-3b44-4012-a793-c6147a4fb1f0","resolution":{"observed_at":"2026-07-10T00:06:38.395137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.401851Z","title":"Automatic generation of software behavioral models,","venue":null,"work_id":"d23b4b59-a8cc-467b-b1eb-64e0dcc0f1d7","year":2008},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:13c3911840361e18f08bc7a86f0f1fd27b6a4196295a06837b8cccd216fb198b","observation_id":"f89dc7ec-340b-4e20-b62f-18d00ca50c54","resolution":{"observed_at":"2026-07-10T00:06:38.403140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.409296Z","title":"Inferring models of concurrent systems from logs of their behavior with CSight,","venue":null,"work_id":"aadaf075-ff44-4bb0-b375-33e081c20de1","year":2014},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:32608b7718e958a46ef95651e62680641336f25ac692b3ff42015d1a2ad11dd3","observation_id":"2dc6c576-bf93-4be9-bcc0-4dc77de44aad","resolution":{"observed_at":"2026-07-10T00:06:38.410430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.410917Z","title":"Workflow mining: Discovering process models from event logs,","venue":null,"work_id":"2770517a-2aeb-4f3e-9d0e-a337b517af4d","year":2004},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:925967bdcd307fbb559afc30990d3907497c1de5a8caeb54ebcc75b55146c41e","observation_id":"a0c731b5-5311-457b-9da7-7e14474b33d1","resolution":{"observed_at":"2026-07-10T00:06:38.412159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.424592Z","title":"Discov- ering block-structured process models from event logs—a constructive approach,","venue":null,"work_id":"469031d6-dc71-46a9-890e-a8a42ba9871d","year":2013},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:5f3469acc951eb6980f234ca913b6546e34b6cc147e741c1072e8a7aec5a4ca7","observation_id":"18aa5651-0695-48e9-822c-646501335467","resolution":{"observed_at":"2026-07-10T00:06:38.425884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.405837Z","title":"Applying graph reduction techniques for identifying structural conflicts in process models,","venue":null,"work_id":"07b30777-4717-4a60-8149-1d1ecf00f785","year":1999},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:e71b31215a894c0f2ec338bdc98ece2175735b854bd0ebaf4ada01b2f8673909","observation_id":"1f2e5127-a857-461b-abc0-1219399aa20c","resolution":{"observed_at":"2026-07-10T00:06:38.407049Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.430274Z","title":"Learning regular sets from queries and counterexamples,","venue":null,"work_id":"f7272fe8-f42d-4fec-b465-11c9ab04ade2","year":1987},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a4b9d761245362c4afbec9d931a47a62479b7bb20dcef939b9f79e942b34c702","observation_id":"d81ef0ce-87de-4eb1-a2a2-1871b5a90b68","resolution":{"observed_at":"2026-07-10T00:06:38.431693Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.414223Z","title":"Unsupervised dialog structure learning,","venue":null,"work_id":"ee07b3c5-8519-4d36-bf8b-6cd33ad7ad8e","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:a3857404fb73ddda9b80bcea7893047e3eac23c20ba827f97fa2f1fb0759d158","observation_id":"9e496994-d97f-432f-b6da-538d382c9579","resolution":{"observed_at":"2026-07-10T00:06:38.415418Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.391907Z","title":"Dialog2Flow: Pre-training soft-contrastive action-driven sentence embeddings for automatic dia- log flow extraction,","venue":null,"work_id":"13746e4c-101b-4b43-a6fe-49f17d2b4ed8","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:b56c89c3f9f1315bca22047e05759e85bed2370be5097f108b7036339c2450bc","observation_id":"8b21782c-e4f6-4cf1-8c76-f9e32d4497df","resolution":{"observed_at":"2026-07-10T00:06:38.393236Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.403829Z","title":"Agenda-based user simulation for bootstrapping a POMDP dialogue system,","venue":null,"work_id":"31398511-6caa-4c5e-bb5d-16330d6159ae","year":2007},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:48b2a471afa672ed225ed64d148abd669985b26281e856d68e6f17aae6973a83","observation_id":"47d02b58-9eb9-4c9d-8561-ca85fee5e5e5","resolution":{"observed_at":"2026-07-10T00:06:38.404999Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.405993Z","title":"ConvLab-2: An open-source toolkit for building, evaluating, and diagnosing dialogue systems,","venue":null,"work_id":"9f5a3aa7-3570-4098-8afe-25b387d95c5a","year":2020},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:e03d4e4d7c70f83805663f4eaba66197b6a529e7fe1c5d25bbe757176d63fd64","observation_id":"8b43514d-7a55-415b-94e7-19673430fb4d","resolution":{"observed_at":"2026-07-10T00:06:38.407241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.384895Z","title":"A taxonomy of model- based testing approaches,","venue":null,"work_id":"54c4e6b2-2fbf-4590-87a1-07218c9cd816","year":2012},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:921289927cb38a8ad5f3a1d86451f89cd4c7492c2661209b030f4d0bb0c49696","observation_id":"2fc55fc5-d849-4b04-aa23-93c0f960bcdb","resolution":{"observed_at":"2026-07-10T00:06:38.385993Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.388516Z","title":"Principles and methods of testing finite state machines—a survey,","venue":null,"work_id":"73b5409a-ac05-46ca-b338-2463c67bb34c","year":1996},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:be19e47efd98c498ad2ba5842d23bf0240d813fdbcbf793875c2fe3e7e268062","observation_id":"75ce28d8-4ae0-40ca-ba6d-2e5f3c4d9ca9","resolution":{"observed_at":"2026-07-10T00:06:38.389665Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.395511Z","title":"Testing software design modeled by finite-state machines,","venue":null,"work_id":"83c208f5-69cf-482a-a3b5-cde6e4e39600","year":1978},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:115190deba8c5331e336ca1d6d13b701248d77177ab18deeb7b49e118da52482","observation_id":"f73ca176-6c76-43e6-9d34-280c4bf33ab5","resolution":{"observed_at":"2026-07-10T00:06:38.396696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.383213Z","title":"RESTler: Stateful REST API fuzzing,","venue":null,"work_id":"5b793712-cb5e-41e5-a09f-1ac09f49ba0e","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3bbac655fca691aa9fd2dfa86af72c663508f6de95aa37796128d944ccb00051","observation_id":"25e0e7b8-c0d7-4dfd-bdc5-0355e81b625d","resolution":{"observed_at":"2026-07-10T00:06:38.384367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.372714Z","title":"RESTful API automated test case generation with Evo- Master,","venue":null,"work_id":"c92a652e-731c-42a0-9221-eaa4a72ee2ae","year":2019},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:02f0a6fdabd9eab81abe861b2e8bfb4d9d4f42cd7f816fe3fbcdd57d80b945fc","observation_id":"413c8bf1-30d6-41dd-9201-7823baa165db","resolution":{"observed_at":"2026-07-10T00:06:38.373884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.349858Z","title":"Morest: Model-based RESTful API testing with execution feedback,","venue":null,"work_id":"5dce314c-5a54-4d89-9e5a-582cb97c2608","year":2022},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:658fd90f5b7eeb3d1dd3ae960b80293c975ce853745ccfc73e061742bbefc612","observation_id":"9c71df55-6f0d-40e5-86cb-c486edd2c362","resolution":{"observed_at":"2026-07-10T00:06:38.351133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.374445Z","title":"KAT: Dependency-aware automated API testing with large language models,","venue":null,"work_id":"b2bd4ef3-dedc-4765-80e8-708ad6ef1753","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:fff8df10f63849b4addb6bfbe9c0fa710c6de431b3ac8f1dd61096805c2721c5","observation_id":"fdfc078e-4aac-477b-a422-f5bbcb06dd86","resolution":{"observed_at":"2026-07-10T00:06:38.375692Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.407771Z","title":"Testing RESTful APIs: A survey,","venue":null,"work_id":"fea21a97-42a0-4ae8-92b2-71721e7196ea","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:9947e693d27991c7c7e9c99e0ba9d8651d4e7c58eb79264321b00523c4c4fc60","observation_id":"5ccd9764-a37b-474b-b12d-7afdcb0079dd","resolution":{"observed_at":"2026-07-10T00:06:38.408976Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.411221Z","title":"An empirical evaluation of using large language models for automated unit test generation,","venue":null,"work_id":"a22a948e-2c9d-4487-8f40-e073a1f67e91","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:3efb32067ef1a4d78c19920464501cf8ed8555d7884f9909eb167d72942e8046","observation_id":"2b3b289f-f2e8-4e78-98d5-f07c0cd671c2","resolution":{"observed_at":"2026-07-10T00:06:38.412345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.416353Z","title":"CoverUp: Effective high coverage test generation for Python,","venue":null,"work_id":"a5bf595d-c2b8-4f43-aea9-ea638da11615","year":2025},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:b96ea19d065703140c6c48f5bafa36a818a9dd372fd59102103cf83fa1f22fb6","observation_id":"5352081e-274f-4b47-bb01-3889218a77e5","resolution":{"observed_at":"2026-07-10T00:06:38.417481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.415993Z","title":"Evaluating and improving ChatGPT for unit test generation,","venue":null,"work_id":"6034b27e-e14c-4e73-90a4-75a4a58f8d21","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:554ec2fe9bf18bf1b4e5326cdb661bce26559859681f819694e3f658f8281880","observation_id":"a19faacc-39ec-42e5-9c38-01a65f71e8cd","resolution":{"observed_at":"2026-07-10T00:06:38.417180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.409474Z","title":"CodaMosa: Escaping coverage plateaus in test generation with pre-trained large language models,","venue":null,"work_id":"aae2f8f5-47e7-4723-ba40-ac50b3eafe98","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:2b3d4c17c36086941d6d414f0b55ea3675c2dadd9395052b60fdab5880d7dc83","observation_id":"c6508104-bd2e-4132-b3b5-e0026d11b7f7","resolution":{"observed_at":"2026-07-10T00:06:38.410662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03095","last_updated":"2025-03-31T13:13:27Z","snapshot_observed_at":"2026-08-20T10:06:07.933389Z","submitted_at":"2024-08-06T10:52:41Z","title":"TestART: Improving LLM-based Unit Testing via Co-evolution of Automated Generation and Repair Iteration","version":6},"cited_work":{"arxiv_id":"2408.03095","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.03095","snapshot_observed_at":"2026-07-10T00:06:37.968483Z","title":"Improving llm-based unit test generation via template-based repair","venue":"cs.SE","work_id":"14eda6da-904c-4e2a-af93-d226b0281ec6","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2408.03095","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:be7e95319edcb0712f4d0b2d1f455af962e833d69a10f5d3e03bbd68d6ea21ce","observation_id":"74a8218e-fa88-4365-9149-1b2c1b93a272","resolution":{"observed_at":"2026-07-10T00:06:37.970137Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.361453Z","title":"The oracle problem in software testing: A survey","venue":null,"work_id":"fd9c3fd4-a9a0-4be7-8284-ff0a990a96f9","year":2015},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:0112df785698c602ad577480ac7afd6c1b245cdc8c112cec90e6e4cd215527dc","observation_id":"d2ee0e6d-23e2-41d1-865d-8836ec6aaeef","resolution":{"observed_at":"2026-07-10T00:06:38.362723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.367375Z","title":"Pseudo-oracles for non-testable programs,","venue":null,"work_id":"d502e25a-8488-41d6-85a0-45321bf83977","year":1981},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:55ad1ad67c579e7bc645c980622c6990f22d740c3a771f9d69df0690e1c5c787","observation_id":"91b24357-a3aa-40bb-a95e-781f191ad93b","resolution":{"observed_at":"2026-07-10T00:06:38.368635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.355218Z","title":"On testing non-testable programs,","venue":null,"work_id":"1c9fc5bb-0e20-4d93-8641-8c567d31df2c","year":1982},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:fe82ed785aa4719139510812d5c76e078588be055f0cfe6c9c4402d6af4387ec","observation_id":"e755b481-bc90-4bad-9642-d31ac5d8b987","resolution":{"observed_at":"2026-07-10T00:06:38.356276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.377998Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":null,"work_id":"cccce1f9-7736-4f3d-8edd-e8144f4dd4b0","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:1dcc97bb9481c5ca49caa1f4a5054dd7fee9a85e103f863ad773925f0ce95939","observation_id":"5280139e-b263-4a46-9432-063d6ccc5e10","resolution":{"observed_at":"2026-07-10T00:06:38.379148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.414605Z","title":"Large language models cannot self-correct reasoning yet","venue":null,"work_id":"8ec9480a-3243-490b-90b3-69edf7b7205a","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:6bf1642649c5dfedd0fac6a52ed4d90c12afe15effc1227f33a050fdaf9e65c5","observation_id":"8919763f-e676-4082-b6e1-3691f34f621c","resolution":{"observed_at":"2026-07-10T00:06:38.415797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08115","last_updated":"2024-08-03T21:25:31Z","snapshot_observed_at":"2026-08-17T20:35:20.009398Z","submitted_at":"2024-02-12T23:11:01Z","title":"On the Self-Verification Limitations of Large Language Models on Reasoning and Planning Tasks","version":2},"cited_work":{"arxiv_id":"2402.08115","doi":"10.48550/arxiv.2402.08115","metadata_source":"pith","pith_arxiv_id":"2402.08115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"On the self-verification limitations of large language models on reasoning and planning tasks","venue":"cs.AI","work_id":"0e539fa7-dc00-479f-bafa-8bda539f2f5d","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"cited_paper":"/paper/2402.08115","citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c6d17684e291e0123063dcf7b94b3c5d524a887772007e05c6e9aa6c5d4d02ba","observation_id":"56f5958e-caa8-4320-9086-ead614e1fd1c","resolution":{"observed_at":"2026-07-10T00:06:37.970034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.370968Z","title":"Self-Refine: Iterative refinement with self-feedback,","venue":null,"work_id":"769df8fb-dbf6-4ea5-8b1f-028ef73398aa","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:de1fbb95562561de83a4c86f30c0400d001ec3171f4da33f4f9033284a52a41e","observation_id":"fc6e1223-e3df-467e-90f8-89dcda18765c","resolution":{"observed_at":"2026-07-10T00:06:38.372182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.376195Z","title":"CRITIC: Large language models can self-correct with tool-interactive critiquing,","venue":null,"work_id":"e512635d-48a0-4eb7-8be0-ac15d2b8a660","year":2024},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:4a186032c9b0769859e768bf43b9df84a827b9728b44f527431c7048ea023d88","observation_id":"ffd8ae57-7bb3-41f1-a318-d4876be7b2df","resolution":{"observed_at":"2026-07-10T00:06:38.377391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.397556Z","title":"Metamorphic testing: A review of challenges and opportunities,","venue":null,"work_id":"fbb46226-564b-4d46-aaec-881b695f7b10","year":2018},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:e8f0ccaed5e69ef192dd55374b511dc91096b13acdce427458a1efdbe5c922c3","observation_id":"4b153557-3da1-4ffe-b1b5-bd2cff717884","resolution":{"observed_at":"2026-07-10T00:06:38.398672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.401698Z","title":"Not what you’ve signed up for: Compromising real-world LLM- integrated applications with indirect prompt injection,","venue":null,"work_id":"4a1d50d5-4caf-46e9-86d9-f2fc58b5b7c0","year":2023},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:c438c86ac5c19a6100fe02f62b993d52fb890b4daa150f658d8d4b85b25f5434","observation_id":"45b02c00-7cdd-4acc-b72b-106b5381a916","resolution":{"observed_at":"2026-07-10T00:06:38.403085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T00:06:38.417794Z","title":"Ignore previous prompt: Attack techniques for language models","venue":null,"work_id":"e1fb2557-e604-461b-8d64-d005e35c230b","year":2022},"citing_paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-07-10T00:02:58.383840Z"},"links":{"citing_paper":"/paper/2607.06873"},"observation_digest":"sha256:fa3f252e2a5ad715b32038f3a0e079c562ac3100548e74206a4c74a4c5884de0","observation_id":"587a72f6-11a9-45de-b57e-fa05e453fc52","resolution":{"observed_at":"2026-07-10T00:06:38.419028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.06873","last_updated":"2026-07-08T00:11:41Z","latest_version":1,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-12T20:36:39.312308Z","submitted_at":"2026-07-08T00:11:41Z","title":"Mining Workflow Graphs for Black-Box Boundary Testing of Conversational LLM Agents"},"reference_resolution":{"displayed":62,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":2,"verified_exact":9,"verified_fuzzy":50},"total_outbound_references":62},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 62 of 62 outbound references and 0 inbound Pith citation observations for arXiv:2607.06873."}