{"as_of":"2026-08-09T15:47:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9f1e8fc90ea788e3fe4fade720d3e4a2f9527e3380e56c661b1b5944cd826087","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:55:55.302080Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-21T02:08:06.976461Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T02:09:24.310163Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"cited_work":{"arxiv_id":"2506.23049","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23049","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aura: Agent for understanding, reasoning, and auto- mated tool use in voice-driven tasks","venue":null,"work_id":"08ca7547-c4d0-48a5-b956-cf3060d2627d","year":2025},"citing_paper":{"arxiv_id":"2604.15037","last_updated":"2026-05-02T12:33:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-16T14:06:30Z","title":"From Reactive to Proactive: Assessing the Proactivity of Voice Agents via ProVoice-Bench","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T11:44:12.373082Z"},"links":{"cited_paper":"/paper/2506.23049","citing_paper":"/paper/2604.15037"},"observation_digest":"sha256:3acdf135e136091b7cfad4a84efcba19a26ab140a38fce96d80329a115156af6","observation_id":"9231afea-d3b6-4e49-9b70-35d3f6ae2021","resolution":{"observed_at":"2026-05-10T11:45:21.002818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"cited_work":{"arxiv_id":"2506.23049","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23049","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aura: Agent for understanding, reasoning, and auto- mated tool use in voice-driven tasks","venue":null,"work_id":"08ca7547-c4d0-48a5-b956-cf3060d2627d","year":2025},"citing_paper":{"arxiv_id":"2605.21008","last_updated":"2026-05-20T10:44:56Z","snapshot_observed_at":"2026-07-06T23:31:32.577242Z","submitted_at":"2026-05-20T10:44:56Z","title":"A Survey of Audio Reasoning in Multimodal Foundation Models","version":1},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-05-21T02:08:06.976461Z"},"links":{"cited_paper":"/paper/2506.23049","citing_paper":"/paper/2605.21008"},"observation_digest":"sha256:4c3b5dbb18262857f5e4d73e2004899be3a3fce6974820f621841bdacfd95db9","observation_id":"878f8158-8b47-4503-8238-062795f7805e","resolution":{"observed_at":"2026-05-21T02:09:24.313261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23049/citation-record","integrity":"/paper/2506.23049/integrity","json":"/paper/2506.23049/citation-record.json","paper":"/paper/2506.23049"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:57.292892Z","title":"Espnet-sds: Unified toolkit and demo for spoken dialogue systems,","venue":null,"work_id":"fe156377-f1a5-451b-9309-de0e20fb9987","year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.060791Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:ad06e7a6a1ab19e93e96ce5522806d41a204fdd5efd66d011c5f0b2be9740503","observation_id":"6a050def-1168-4fe9-8295-ecffce12c92a","resolution":{"observed_at":"2026-08-06T21:55:57.345167Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11000","last_updated":"2023-05-19T14:41:16Z","snapshot_observed_at":"2026-08-07T10:56:05.622094Z","submitted_at":"2023-05-18T14:23:25Z","title":"SpeechGPT: Empowering Large Language Models with Intrinsic Cross-Modal Conversational Abilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11000","snapshot_observed_at":"2026-08-06T21:55:53.134838Z","title":"Speechgpt: Empowering large language models with intrinsic cross-modal conversational abilities,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.134838Z"},"links":{"cited_paper":"/paper/2305.11000","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:2e44e0ef94693633b1709a3b55faf4395c506279bd52a6e9f547f2c7b124c7b6","observation_id":"0dc7d8ef-a7bb-4d91-b572-23ebe7d1ba76","resolution":{"observed_at":"2026-08-06T21:55:53.134838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17239","last_updated":"2025-02-24T15:16:34Z","snapshot_observed_at":"2026-08-07T17:53:08.473463Z","submitted_at":"2025-02-24T15:16:34Z","title":"Baichuan-Audio: A Unified Framework for End-to-End Speech Interaction","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.17239","snapshot_observed_at":"2026-08-06T21:55:53.227261Z","title":"Baichuan- audio: A unified framework for end-to-end speech interaction,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.227261Z"},"links":{"cited_paper":"/paper/2502.17239","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:beacafa58037ff74d3b41256153bff52f5ea68b25f36ef3794dda25cae9eefad","observation_id":"be26c935-a7ac-46cc-8640-094e2ff6b65b","resolution":{"observed_at":"2026-08-06T21:55:53.227261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16725","last_updated":"2024-11-05T02:24:18Z","snapshot_observed_at":"2026-07-06T19:07:46.545514Z","submitted_at":"2024-08-29T17:18:53Z","title":"Mini-Omni: Language Models Can Hear, Talk While Thinking in Streaming","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.16725","snapshot_observed_at":"2026-08-06T21:55:53.426594Z","title":"Mini-omni: Language models can hear, talk while thinking in streaming,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.426594Z"},"links":{"cited_paper":"/paper/2408.16725","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:422bbba3a00324cc45baaf17210a448f92fbbf8549a01041ae43b4dfe2e90c61","observation_id":"83cfe486-8963-4984-b578-6021b9fedb3c","resolution":{"observed_at":"2026-08-06T21:55:53.426594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:53.535944Z","title":"Openomni: Advancing open-source omnimodal large language models with progressive multimodal alignment and real-time self-aware emotional speech synthesis,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.535944Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:ceb34b51324684d4b7bedb375b835d8975403ccd405e604d6a6e39e4cc8882f5","observation_id":"6c4f318d-5e12-4868-9bfc-5745befd587c","resolution":{"observed_at":"2026-08-06T21:55:53.535944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17196","last_updated":"2024-12-11T15:45:21Z","snapshot_observed_at":"2026-07-06T19:37:56.214143Z","submitted_at":"2024-10-22T17:15:20Z","title":"VoiceBench: Benchmarking LLM-Based Voice Assistants","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17196","snapshot_observed_at":"2026-08-06T21:55:53.632465Z","title":"V oicebench: Benchmarking llm-based voice assistants,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.632465Z"},"links":{"cited_paper":"/paper/2410.17196","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:a966f61b01d58abceb4c482dcc3edc1da7b17e16335bce1ae7aadfefe05d3f64","observation_id":"29136298-6744-4ed0-ad02-6f8d95d881cf","resolution":{"observed_at":"2026-08-06T21:55:53.632465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.00278","last_updated":"2020-04-20T15:02:43Z","snapshot_observed_at":"2026-08-09T13:11:09.932297Z","submitted_at":"2018-09-29T23:44:39Z","title":"MultiWOZ -- A Large-Scale Multi-Domain Wizard-of-Oz Dataset for Task-Oriented Dialogue Modelling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.00278","snapshot_observed_at":"2026-08-06T21:55:53.745815Z","title":"Multiwoz–a large-scale multi-domain wizard- of-oz dataset for task-oriented dialogue modelling,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.745815Z"},"links":{"cited_paper":"/paper/1810.00278","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:85a54c8a197ece0b5c5f302e5825587809cd1f24446e9b467c3531dacf0d6770","observation_id":"0a0bb2e1-e7cd-43f1-8433-9cbfe2ada353","resolution":{"observed_at":"2026-08-06T21:55:53.745815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13040","last_updated":"2025-06-24T07:06:57Z","snapshot_observed_at":"2026-07-06T15:30:42.173342Z","submitted_at":"2023-05-22T13:47:51Z","title":"SpokenWOZ: A Large-Scale Speech-Text Benchmark for Spoken Task-Oriented Dialogue Agents","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13040","snapshot_observed_at":"2026-08-06T21:55:53.858082Z","title":"Spokenwoz: A large-scale speech-text benchmark for spoken task-oriented dialogue agents,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.858082Z"},"links":{"cited_paper":"/paper/2305.13040","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:606cd7de0ffc0b00415a887f413af972ea432c2e6826dd0e226dde6798058161","observation_id":"1729b4ea-7e13-4fd2-8e35-6e6537c543bc","resolution":{"observed_at":"2026-08-06T21:55:53.858082Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-06T21:55:53.956373Z","title":"Training language models to follow instructions with human feedback,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.956373Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:2717a1b0ffc7f89c1e2cc36fcbf8f0de3da62802713bca4c626a5f322ea1bc1c","observation_id":"1b3c17e9-60ec-4a67-93ca-76db04cd35ad","resolution":{"observed_at":"2026-08-06T21:55:53.956373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17580","last_updated":"2023-12-03T18:17:21Z","snapshot_observed_at":"2026-08-03T00:52:54.308486Z","submitted_at":"2023-03-30T17:48:28Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17580","snapshot_observed_at":"2026-08-06T21:55:54.050660Z","title":"Hugginggpt: Solving ai tasks with chatgpt and its friends in hugging face,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.050660Z"},"links":{"cited_paper":"/paper/2303.17580","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:f77c150a7bd468a4d1466ca173ba2532cb752153c57789eea0bbb9f90c842305","observation_id":"556260fb-0aa7-484a-98ae-0209d064c99e","resolution":{"observed_at":"2026-08-06T21:55:54.050660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T21:55:54.130776Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.130776Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:61b26a3b3723aac12f66898c2e2c8182131e088853b037b1976c1c66a8550a7a","observation_id":"5e8bd2b4-b095-4eaa-96f2-ef35ab2f289f","resolution":{"observed_at":"2026-08-06T21:55:54.130776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.00564","last_updated":"2025-03-01T17:23:51Z","snapshot_observed_at":"2026-08-07T17:37:04.045007Z","submitted_at":"2025-03-01T17:23:51Z","title":"ToolDial: Multi-turn Dialogue Generation Method for Tool-Augmented Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.00564","snapshot_observed_at":"2026-08-06T21:55:54.171642Z","title":"Tooldial: Multi-turn dialogue generation method for tool-augmented language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.171642Z"},"links":{"cited_paper":"/paper/2503.00564","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:c262edd69e100f558ae58476ec886ad61b376bae136d54f754bb29a4c4d6e412","observation_id":"44b5b849-70de-40c6-b1f0-aaa0cafb8fa0","resolution":{"observed_at":"2026-08-06T21:55:54.171642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:57.197559Z","title":"Rethinking task-oriented dialogue systems: From complex modularity to zero-shot autonomous agent,","venue":null,"work_id":"c1a902f3-f777-4618-a4b3-727cccdf3660","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.215277Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:a6f39f7de17fa97954a34e626b83b6d038cc9b6119072a3bd0a8143b0b822a95","observation_id":"6fb7d218-242a-4f0f-af21-6204b876b14c","resolution":{"observed_at":"2026-08-06T21:55:57.237848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:54.285113Z","title":"React: Synergizing reasoning and acting in language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.285113Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:9253fe5ed8e54312e58ace235faa464e59dbcaae583b9649b46c7500a058a2ce","observation_id":"a9230911-5ef5-4918-bd37-7749709b1904","resolution":{"observed_at":"2026-08-06T21:55:54.285113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:57.098888Z","title":"Audio-cot: Exploring chain-of-thought reasoning in large audio language model,","venue":null,"work_id":"be34954d-1e08-4534-8a9b-e65237e3e07f","year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.342505Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:e1be315c9ab74532d1504669787d16ba9d2bff88e7d0fdfb1ee2a24b05407c3b","observation_id":"899075e1-9ab3-44f9-bedd-970ffb07a5f2","resolution":{"observed_at":"2026-08-06T21:55:57.139353Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:54.436315Z","title":"Audio-reasoner: Improving reasoning capability in large audio language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.436315Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:6bca7335b631e30f9af5a2716cdd67d8019d51bc3f0909c45198a9ce6f732eed","observation_id":"75024118-a0c6-4ac4-b6bf-d45f86f1d41e","resolution":{"observed_at":"2026-08-06T21:55:54.436315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07246","last_updated":"2025-01-13T11:54:40Z","snapshot_observed_at":"2026-08-07T04:02:14.954506Z","submitted_at":"2025-01-13T11:54:40Z","title":"Audio-CoT: Exploring Chain-of-Thought Reasoning in Large Audio Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07246","snapshot_observed_at":"2026-08-06T21:55:54.386404Z","title":"Available: https://arxiv.org/abs/2501.07246","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.386404Z"},"links":{"cited_paper":"/paper/2501.07246","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:1c828326ae50fb5555e202a4e0375ab83a819e754c6be37689a89fd6437770e0","observation_id":"13a23776-4513-41db-ab68-e23694a70cd5","resolution":{"observed_at":"2026-08-06T21:55:54.386404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:54.539532Z","title":"Can a suit of armor conduct electricity? a new dataset for open book question answering,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.539532Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:8aa53aa721095f598adf489e50cf9ddcfd270a51f7161749b0021045f843c774","observation_id":"5f6776fd-53f7-48e7-a592-eb9359226dfc","resolution":{"observed_at":"2026-08-06T21:55:54.539532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00927","last_updated":"2025-04-19T15:43:36Z","snapshot_observed_at":"2026-08-04T03:00:55.235335Z","submitted_at":"2024-11-01T15:57:45Z","title":"ReSpAct: Harmonizing Reasoning, Speaking, and Acting Towards Building Large Language Model-Based Conversational AI Agents","version":2},"cited_work":{"arxiv_id":"2411.00927","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.00927","snapshot_observed_at":"2026-08-06T21:55:55.409227Z","title":"ReSpAct: Harmonizing Reasoning, Speaking, and Acting Towards Building Large Language Model-Based Conversational AI Agents","venue":"cs.CL","work_id":"e7a3a49d-06db-411d-a014-44b3425ab948","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.473272Z"},"links":{"cited_paper":"/paper/2411.00927","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:d8fa25e65e534be64d0cee72bae6acd41b9600ae32f1793750d7bf5baa82de4a","observation_id":"e08f5fdd-a684-4b24-9a06-34a544179e27","resolution":{"observed_at":"2026-08-06T21:55:55.474892Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.986840Z","title":"Gpt-4o: Openai’s new multimodal flagship model,","venue":null,"work_id":"ab951c5c-196c-47c3-bf34-8d3c017c42a9","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.625379Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:8ff387a04f01e9911f230121a1133209136b522dd4dd289774a5fbb9ccf48748","observation_id":"11404fd2-5124-4b62-a298-96fbd8fde36c","resolution":{"observed_at":"2026-08-06T21:55:57.037215Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.04356","last_updated":"2022-12-06T18:46:04Z","snapshot_observed_at":"2026-07-06T14:28:21.844826Z","submitted_at":"2022-12-06T18:46:04Z","title":"Robust Speech Recognition via Large-Scale Weak Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.04356","snapshot_observed_at":"2026-08-06T21:55:54.575759Z","title":"Robust speech recognition via large-scale weak supervision,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.575759Z"},"links":{"cited_paper":"/paper/2212.04356","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:05ce831614723014ea42faf4b200eb11be37c832c606c5313bd5d8bd938f88a6","observation_id":"13a632cc-e96a-46d8-820f-e45d952320fc","resolution":{"observed_at":"2026-08-06T21:55:54.575759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.687099Z","title":"Espnet-TTS: Unified, reproducible, and integratable open source end-to-end text-to-speech toolkit,","venue":null,"work_id":"d1c4bf4a-1b00-4022-b81a-69ea94421bfb","year":2020},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.714602Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:0c5b8dc903db267879dac203fe959e38c917a1d501ee67564ac4e77808b1cb57","observation_id":"3705390a-447d-4e6f-b7d1-6b65e8ad02ce","resolution":{"observed_at":"2026-08-06T21:55:56.795127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.859364Z","title":"Owsm v3.1: Better and faster open whisper-style speech models based on e-branchformer,","venue":null,"work_id":"4f66d0ea-16f0-430c-9db2-a78208c6e42b","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.650824Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:6c6be49e422a3f588591b9b28337fa4113d93757e19d1bb9546ea154b3f4eed6","observation_id":"37471f67-73b0-461a-8515-e273a18636a3","resolution":{"observed_at":"2026-08-06T21:55:56.937871Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-06T21:55:54.808526Z","title":"Efficient memory management for large language model serving with pagedattention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.808526Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:8247840151e6d45cd50b9c065251f8d9b2e754cc89da27eef5b52e2b6ad409f9","observation_id":"66ba6451-65ec-41fd-9e2c-1ef0c70223ec","resolution":{"observed_at":"2026-08-06T21:55:54.808526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T21:55:54.758002Z","title":"The llama 3 herd of models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.758002Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:f057830933b6d8de15d622562a85c4d1bc1709b15ae2a8e747bc79037deaf0db","observation_id":"4d4e4b70-79e5-4f2e-bca8-d5e1e66c4115","resolution":{"observed_at":"2026-08-06T21:55:54.758002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.327294Z","title":"Gpt-4o: Openai’s new multimodal flagship model,","venue":null,"work_id":"289e120d-db94-4e5c-9587-4e21da7e2fc0","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.920074Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:408156382d191fc14869fe7dfc18d6dd840b52ca6c973e24d975ed866186fd77","observation_id":"c741c3e5-60be-4f0c-a95b-f7eb3d2c3196","resolution":{"observed_at":"2026-08-06T21:55:56.399442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.474970Z","title":"Alpacaeval: An automatic evaluator of instruction-following models,","venue":null,"work_id":"94e4882c-6b69-4577-813b-8c3160cacae5","year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.850695Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:867aa1cd4546ab636b8b6ba221c3edc3cf648adc5b088ce0426160ff1884d7d7","observation_id":"64c3b58c-b427-4d79-946b-af719df12293","resolution":{"observed_at":"2026-08-06T21:55:56.544396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.00037","last_updated":"2024-10-02T09:11:45Z","snapshot_observed_at":"2026-07-30T10:21:14.474746Z","submitted_at":"2024-09-17T17:55:39Z","title":"Moshi: a speech-text foundation model for real-time dialogue","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.00037","snapshot_observed_at":"2026-08-06T21:55:55.026534Z","title":"Moshi: a speech-text foundation model for real-time dialogue,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.026534Z"},"links":{"cited_paper":"/paper/2410.00037","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:5bf00f051dc338a4c21ce9ad4cbe6fcf09b9d5a4faa4cbdf0a18a93d03c2aa20","observation_id":"4c43f8dc-f9de-4f62-8a4b-20a10ca0f237","resolution":{"observed_at":"2026-08-06T21:55:55.026534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:56.121062Z","title":"Gpt-4o: Openai’s new multimodal flagship model,","venue":null,"work_id":"e972c10a-c454-4d73-8882-5d971d9fa652","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.953118Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:7abcfa3ade056fe554e251c951a50f86726661d65eedc9bf0557c9b9aaa60c92","observation_id":"32a86f2f-3495-438e-ad5a-6029d8f42bd6","resolution":{"observed_at":"2026-08-06T21:55:56.224799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18425","last_updated":"2025-04-25T15:31:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-25T15:31:46Z","title":"Kimi-Audio Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.18425","snapshot_observed_at":"2026-08-06T21:55:55.166303Z","title":"Kimi-audio technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.166303Z"},"links":{"cited_paper":"/paper/2504.18425","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:80b8d8f95884e7e93629fdcb5ebf373d1a4f597363b999136d2a44832edd9e3b","observation_id":"f6d58d67-3f07-4053-a581-dfd13108f015","resolution":{"observed_at":"2026-08-06T21:55:55.166303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11190","last_updated":"2024-11-05T02:27:57Z","snapshot_observed_at":"2026-07-06T19:33:34.819837Z","submitted_at":"2024-10-15T02:10:45Z","title":"Mini-Omni2: Towards Open-source GPT-4o with Vision, Speech and Duplex Capabilities","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11190","snapshot_observed_at":"2026-08-06T21:55:55.082723Z","title":"Mini-omni2: Towards open-source gpt-4o with vision, speech and duplex capabilities,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.082723Z"},"links":{"cited_paper":"/paper/2410.11190","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:576ce178f5ae9ab77035100dc155ac71b7b8fd4e211fb7f1b56b227ea51e8e6b","observation_id":"63d253aa-d644-45da-a7b8-a64c00d06d94","resolution":{"observed_at":"2026-08-06T21:55:55.082723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:55:55.925500Z","title":"Parakeet-tdt-0.6b-v2,","venue":null,"work_id":"ddfbbe92-1b75-4615-8d55-4fff2f374293","year":2024},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.302080Z"},"links":{"citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:03ed5b52333f860dc878a7e96138fb1058f55744af46adca670d51b1349556f9","observation_id":"342d84ad-921d-4ceb-a81f-bd574f5321f8","resolution":{"observed_at":"2026-08-06T21:55:56.017766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T21:55:55.221863Z","title":"Qwen3 technical report,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:55.221863Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:334936bfd8e03f14f67a8e74e575288ad3abb154279c130c830b92d9a1116dca","observation_id":"7640f52d-8697-41b2-b649-200ececb7083","resolution":{"observed_at":"2026-08-06T21:55:55.221863Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.08533","last_updated":"2025-03-11T15:24:02Z","snapshot_observed_at":"2026-08-07T17:12:39.670272Z","submitted_at":"2025-03-11T15:24:02Z","title":"ESPnet-SDS: Unified Toolkit and Demo for Spoken Dialogue Systems","version":1},"cited_work":{"arxiv_id":"2503.08533","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.08533","snapshot_observed_at":"2026-08-06T21:55:55.772229Z","title":"ESPnet-SDS: Unified Toolkit and Demo for Spoken Dialogue Systems","venue":"cs.CL","work_id":"abdcf874-c917-4d86-b45f-1cf470770a44","year":2025},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:53.100157Z"},"links":{"cited_paper":"/paper/2503.08533","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:d5c8b43e6cf58251d9e5d74d41104d161f8a87d3e13d78724037ba0ae48ca8a1","observation_id":"853e5859-4f46-4486-a17e-4cc553b47172","resolution":{"observed_at":"2026-08-06T21:55:55.822601Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-09T12:48:24.104649Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":22,"verified_exact":1,"verified_fuzzy":10},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 2 inbound Pith citation observations for arXiv:2506.23049."}