{"as_of":"2026-08-10T13:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:69fcce85826c870e36019438c5cccfd3c9cd152cd52c30d9d159f376c112b378","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":8,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":8,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:05:14.197126Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T15:47:23.273387Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-07T15:05:14.197126Z","title":"Speech model pre-training for end-to-end spoken language under- standing,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2505.16369","last_updated":"2025-05-27T06:49:39Z","snapshot_observed_at":"2026-08-09T02:59:05.011197Z","submitted_at":"2025-05-22T08:23:54Z","title":"X-ARES: A Comprehensive Framework for Assessing Audio Encoder Performance","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T15:05:14.197126Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2505.16369"},"observation_digest":"sha256:65a719ae508531e9f34e6a2de9031d0627598bb4b5ec619a018e7c4a35c56d1c","observation_id":"d5e70fa5-877b-462d-8c31-0932f4dfd7eb","resolution":{"observed_at":"2026-08-07T15:05:14.197126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-07T14:30:33.562288Z","title":"Speech model pre-training for end-to-end spoken language understanding,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2505.18644","last_updated":"2025-05-24T11:09:13Z","snapshot_observed_at":"2026-08-09T11:08:24.041213Z","submitted_at":"2025-05-24T11:09:13Z","title":"Enhancing Generalization of Speech Large Language Models with Multi-Task Behavior Imitation and Speech-Text Interleaving","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:30:33.562288Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2505.18644"},"observation_digest":"sha256:37ffe0442159f759df8b99563704b0127c28dea3c7d7fcbca31f49c7ce1c54c7","observation_id":"5cc71a69-6342-4500-8b2d-4106f84f54fa","resolution":{"observed_at":"2026-08-07T14:30:33.562288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-07T11:46:59.294977Z","title":"Speech model pre-training for end-to-end spoken language understanding,","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2506.01496","last_updated":"2025-06-03T10:16:03Z","snapshot_observed_at":"2026-08-10T07:39:24.689367Z","submitted_at":"2025-06-02T09:59:35Z","title":"Continual Speech Learning with Fused Speech Features","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:46:59.294977Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2506.01496"},"observation_digest":"sha256:6fe50d010885ba9a6a0ed1f0ba28b7f633dd5d59762017326c14c08a1dd6ca76","observation_id":"c2240fbd-62e5-4a08-9b63-b53fdd165702","resolution":{"observed_at":"2026-08-07T11:46:59.294977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-07-13T17:37:12.659819Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2603.26292","last_updated":"2026-06-15T19:50:12Z","snapshot_observed_at":"2026-08-08T09:39:10.545296Z","submitted_at":"2026-03-27T11:03:08Z","title":"findsylls: A Language-Agnostic Toolkit for Syllable-Level Speech Tokenization and Embedding","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-13T17:37:12.659819Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2603.26292"},"observation_digest":"sha256:0bea9d86e4e356534f537fc937b5eabf758e4726c6cceac52c09dcaccb305abb","observation_id":"9107ecd9-b34d-4846-b3be-055b6f96949a","resolution":{"observed_at":"2026-07-13T17:37:12.659819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":"1904.03670","doi":null,"metadata_source":"pith","pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-07-10T15:47:23.273387Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","venue":"eess.AS","work_id":"4a03a99b-cf89-4a7e-8861-8e2917fd8ec2","year":2019},"citing_paper":{"arxiv_id":"2606.26556","last_updated":"2026-06-25T03:07:34Z","snapshot_observed_at":"2026-08-07T18:42:28.880610Z","submitted_at":"2026-06-25T03:07:34Z","title":"WQ-Fusion: Dynamic Gated Attention for Cross-Domain Audio Representation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-26T03:21:39.842959Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2606.26556"},"observation_digest":"sha256:26f28afa109171b4aa531286c021cc10fe4b6749f32f3e6962f4eee35fba058f","observation_id":"b66e635a-b0a0-4956-a1ae-7a1e5a142bb2","resolution":{"observed_at":"2026-07-04T14:39:57.212554Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":"1904.03670","doi":null,"metadata_source":"pith","pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-07-10T15:47:23.273387Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","venue":"eess.AS","work_id":"4a03a99b-cf89-4a7e-8861-8e2917fd8ec2","year":2019},"citing_paper":{"arxiv_id":"2607.07907","last_updated":"2026-07-08T20:42:46Z","snapshot_observed_at":"2026-08-07T04:49:20.628249Z","submitted_at":"2026-07-08T20:42:46Z","title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","version":1},"reference_index":249,"source":"arxiv_source","source_observed_at":"2026-07-10T15:38:58.361411Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2607.07907"},"observation_digest":"sha256:2b2888e0c1df097211849f12ac3179567c53b4bd9f03ac9b7232ee8593bfada4","observation_id":"741fbe33-76d3-4287-8649-b56b6537444b","resolution":{"observed_at":"2026-07-10T15:47:23.274542Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-01T01:19:07.656877Z","title":"Speech model pre-training for end-to-end spoken language understanding,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.25870","last_updated":"2026-07-28T15:37:44Z","snapshot_observed_at":"2026-08-06T06:44:04.480037Z","submitted_at":"2026-07-28T15:37:44Z","title":"VAD to the Bone: Ultra-Tiny Speech Activity Detection for Edge Deployment","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-01T01:19:07.656877Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2607.25870"},"observation_digest":"sha256:d08c92b16c27c49e32e553cf103ac9adb4185a8ef695f81e53259d163b020e4e","observation_id":"c69cc7ce-597c-4dab-8348-63faafe4144f","resolution":{"observed_at":"2026-08-01T01:19:07.656877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.03670","snapshot_observed_at":"2026-08-03T08:50:37.709583Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding","venue":null,"work_id":null,"year":1904},"citing_paper":{"arxiv_id":"2607.29353","last_updated":"2026-07-31T12:37:48Z","snapshot_observed_at":"2026-08-05T23:13:16.167684Z","submitted_at":"2026-07-31T12:37:48Z","title":"Versatile On-device Adaptation at the Edge by Unifying Few-shot, Zero-shot, Continual, and In-context Learning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-03T08:50:37.709583Z"},"links":{"cited_paper":"/paper/1904.03670","citing_paper":"/paper/2607.29353"},"observation_digest":"sha256:ef945048542cda015312fcd366c1e629fb6f5751fc0d80eab7dbe97ca1c1250a","observation_id":"f248c76e-2494-4794-a3f7-03a9e0cb401f","resolution":{"observed_at":"2026-08-03T08:50:37.709583Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/1904.03670/citation-record","integrity":"/paper/1904.03670/integrity","json":"/paper/1904.03670/citation-record.json","paper":"/paper/1904.03670"},"outbound":[],"paper":{"arxiv_id":"1904.03670","last_updated":"2019-07-25T17:56:23Z","latest_version":2,"primary_category":"eess.AS","snapshot_observed_at":"2026-07-06T07:44:28.011439Z","submitted_at":"2019-04-07T15:24:32Z","title":"Speech Model Pre-training for End-to-End Spoken Language Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 8 inbound Pith citation observations for arXiv:1904.03670."}