{"as_of":"2026-08-12T08:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:75381b598a5f4f79eaf98fdf8f714fe5b6a504b0a4279e558a5acf84ff011283","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T13:55:19.657537Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T11:50:21.120714Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T13:55:20.441277Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"cited_work":{"arxiv_id":"2509.00186","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.00186","snapshot_observed_at":"2026-08-05T13:55:20.441277Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","venue":"cs.SD","work_id":"cb0884d3-eb96-434a-b710-ca75a3e2e4f5","year":2025},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.355107Z"},"links":{"cited_paper":"/paper/2509.00186","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:3eb73cfdd417f25dba7eea5a683777cc0db8bac2529f5c12c70da44198dcc749","observation_id":"539820be-08c3-4cfc-9e72-11de312d9131","resolution":{"observed_at":"2026-08-05T13:55:20.496504Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.00186","snapshot_observed_at":"2026-08-08T11:50:21.120714Z","title":"arXiv preprint arXiv:2509.00186 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05507","last_updated":"2026-08-06T01:19:35Z","snapshot_observed_at":"2026-08-09T23:10:58.519642Z","submitted_at":"2026-08-06T01:19:35Z","title":"AffectDF: The Most Comprehensive Benchmark for Speech Deepfake Detection against Emotionally Expressive Attacks","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-08T11:50:21.120714Z"},"links":{"cited_paper":"/paper/2509.00186","citing_paper":"/paper/2608.05507"},"observation_digest":"sha256:256ccfce922e636bf1d633e8de3b1c502842a3222427c5868d1b14a529d65315","observation_id":"d3aa8b9d-dd17-4099-b745-ff8498d19f03","resolution":{"observed_at":"2026-08-08T11:50:21.120714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.00186/citation-record","integrity":"/paper/2509.00186/integrity","json":"/paper/2509.00186/citation-record.json","paper":"/paper/2509.00186"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.642283Z","title":"In the Wild","venue":null,"work_id":"f7ba69ad-367a-45b1-b6ed-89222f95b6c8","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.210708Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:1673fc731ecde65c6512cb36d36758d975d05e31ddfe89eadc1ec21dd85e9f1e","observation_id":"69896137-6f92-4866-b7d0-4dd1e8145067","resolution":{"observed_at":"2026-08-05T13:55:25.733276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.459677Z","title":"Initially, the input audio waveform is chunked into frames and each chunk is processed through TRILL or TRILLs- son models to extract audio representations","venue":null,"work_id":"80a92ba1-efe6-4512-a081-77a8bedc650d","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.438087Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:b1c18625c70d8a922cfcfd2bece6738ab7872df6012cfe0f3ed31727557b85d3","observation_id":"24dc9567-986e-41ab-98fc-8fe2e3c9c445","resolution":{"observed_at":"2026-08-05T13:55:25.555041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.281077Z","title":"Datasets W e conducted extensive experiments using four distinct English datasets","venue":null,"work_id":"c5dcc74a-698a-4208-89ca-9e9a395ae80f","year":2019},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.535619Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:3cfa17d2f6d7ffc46b23b7655eec47c3bb67b5068219fd32640808d76a1ccfb6","observation_id":"12dc41d1-7d25-4fa1-b5e9-de1987b5012c","resolution":{"observed_at":"2026-08-05T13:55:25.351331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:25.035556Z","title":"The results are presented in T able 1","venue":null,"work_id":"58dd1849-7726-4d80-abfc-8b01b28ac91e","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.601890Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:963ba9b217607201c890365945cd8e15b44c4d8c9974364e96d1f79cf847e3e8","observation_id":"dbacd509-8b14-47f6-9841-fdaa09f9df58","resolution":{"observed_at":"2026-08-05T13:55:25.138367Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.712983Z","title":"W e perform extensive experiments to find the most suitable TRILLsson model and optimal chunking duration to balance the local and global temporal features","venue":null,"work_id":"f32aa410-5145-4ac7-af76-799445d33640","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.819089Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:f392c9fabe3c34dd4c2b07439bd8d573355bfafd420127385ff8b49fb591021d","observation_id":"52c2db8e-d509-40f9-ab54-82d430059eeb","resolution":{"observed_at":"2026-08-05T13:55:24.776420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.533946Z","title":null,"venue":null,"work_id":"fb256368-602e-4c1d-a6ab-71ed435ff10d","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.895836Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:ed06a4fb97c0d9693c1a8a542baa6ff8627b7d9ef9fa7ec257859cd7e21005e1","observation_id":"e1217572-2282-4a80-82cf-c252e5ab2875","resolution":{"observed_at":"2026-08-05T13:55:24.609030Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.583876Z","title":"Relative phase information for detecting human speech and spoofed speech","venue":null,"work_id":"3fcb3bcf-fcda-4a31-a7b4-a0ef366945bf","year":2015},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.468369Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:f9596cc828caae5fa1d81d3a5214472437e25d0e8bafd314760f3281dd751fc5","observation_id":"03d5bfdc-5248-4e35-975c-9eecea3681c7","resolution":{"observed_at":"2026-08-05T13:55:23.684648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"cited_work":{"arxiv_id":"2509.00186","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.00186","snapshot_observed_at":"2026-08-05T13:55:20.441277Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","venue":"cs.SD","work_id":"cb0884d3-eb96-434a-b710-ca75a3e2e4f5","year":2025},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.355107Z"},"links":{"cited_paper":"/paper/2509.00186","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:3eb73cfdd417f25dba7eea5a683777cc0db8bac2529f5c12c70da44198dcc749","observation_id":"539820be-08c3-4cfc-9e72-11de312d9131","resolution":{"observed_at":"2026-08-05T13:55:20.496504Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.366238Z","title":"ASVspoof 2019: A large-scale public database of synthesized, converted and replayed speech,","venue":null,"work_id":"d090d6c9-e6c4-44d3-bbe6-a371a6f986d0","year":2019},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.971530Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:5a19a6275c04a9f469d9a583a2c331bf97642955a38bb48c1a4725c7d4700413","observation_id":"ad9bf454-b274-4279-9ca1-b4ca27ac1e9c","resolution":{"observed_at":"2026-08-05T13:55:24.436508Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:17.049614Z","title":"Does audio deepfake detection generalize?","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.049614Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:c576307618b35ddc1188b7ad3718ed771ef621095033790d188c544fab9f5e9e","observation_id":"1c65d7c4-6307-432c-a9a0-aeb6b434c17a","resolution":{"observed_at":"2026-08-05T13:55:17.049614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.132691Z","title":"Resnet and model fusion for automatic spoofing detection","venue":null,"work_id":"cd09c863-ee66-4d24-acbe-1ab410d5c25a","year":2017},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.130519Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:eeac154f355d199d1ae28e35772cbc511d44304a23c8d545faee52c91011fadd","observation_id":"e061d810-75a2-480f-a57a-cdce569056ac","resolution":{"observed_at":"2026-08-05T13:55:24.232885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.953319Z","title":"Replay and synthetic speech detection with res2net architecture,","venue":null,"work_id":"992df686-a9ae-42c2-a7c2-d88a56b9d1ef","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.201712Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:0c1855be55d010e9836e05ce5814303928352b583f8f3454669c7c15703dd6d7","observation_id":"67640e66-052f-4740-8443-76367753ba39","resolution":{"observed_at":"2026-08-05T13:55:24.033733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.757064Z","title":"Fake speech detection using residual network with transformer encoder,","venue":null,"work_id":"955dd1f3-2fb4-49b4-af06-79674c299c35","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.276137Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:05ae4ffd9fee48a2dc9fa7a4a63ae32c56f4f37651b2824f887a8b2c6685c333","observation_id":"7c1632a8-6075-4c16-a8aa-95ffd633ceb5","resolution":{"observed_at":"2026-08-05T13:55:23.848346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14970","last_updated":"2023-08-29T01:50:01Z","snapshot_observed_at":"2026-08-11T22:52:59.862348Z","submitted_at":"2023-08-29T01:50:01Z","title":"Audio Deepfake Detection: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14970","snapshot_observed_at":"2026-08-05T13:55:17.373586Z","title":"Audio deepfake detection: A survey,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.373586Z"},"links":{"cited_paper":"/paper/2308.14970","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:b754deda6f0a340f38d889d34f89a8cc9279da7983d93189f56f29e8742b8e26","observation_id":"bfffb3f5-1a15-40f7-8490-85f8460c9173","resolution":{"observed_at":"2026-08-05T13:55:17.373586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:22.189781Z","title":"AASIST: Audio anti-spoofing using integrated spectro-temporal graph attention networks,","venue":null,"work_id":"5ac58a41-eabb-4726-8b00-adb0e63f42d6","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.132951Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:8d6d1000ae943919576a3fdff0c9e88f3d477681a137c99e797c6b3da3c37b38","observation_id":"78f7892f-3096-407b-a526-7d0f98aa1e50","resolution":{"observed_at":"2026-08-05T13:55:22.327457Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2107.12018","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.183514Z","title":"Ur channel-robust synthetic speech detection system for asvspoof 2021,","venue":null,"work_id":"e483e3ab-f0ce-4740-ace0-8bb6513f4acb","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.535922Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:8503828a62e8d7877c4780894040ccbef7c31294893f5dcb2336715b4c07c7ac","observation_id":"acb983c4-7bea-41d7-914e-729dabfc3ecf","resolution":{"observed_at":"2026-08-05T13:55:20.240050Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.358186Z","title":"A comparison of features for synthetic speech detection,","venue":null,"work_id":"985a810e-253e-491d-a61e-a419a7366b2c","year":2015},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.671294Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:6a4c131de3d2d8f899e3bf87936f086740d5652b3b8e4a597af103f0610e6f5a","observation_id":"82df6cba-b3bb-4d3f-98a8-af410fd48480","resolution":{"observed_at":"2026-08-05T13:55:23.510119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:23.084331Z","title":"T owards end-to-end synthetic speech detection,","venue":null,"work_id":"911c99e5-97c6-4cc0-9f78-324b31123b7b","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.729282Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:81825101e637e2c4b221c81660ece7e95fa11e8fde2bf8edd9e7031b9ba2330d","observation_id":"9b35c735-b1f7-4ede-af1e-003e90059020","resolution":{"observed_at":"2026-08-05T13:55:23.181296Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:22.754113Z","title":"Multi-task learning improves synthetic speech detection,","venue":null,"work_id":"6960c8dd-46e5-4615-893d-780394fbd59c","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.821025Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:3dc9d0b19935a2bf31a426d9a13ecf76507a499436a800628f61ece1a9d8c5fc","observation_id":"9bb41311-c237-4418-a328-f622d693e473","resolution":{"observed_at":"2026-08-05T13:55:22.906864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.12710","last_updated":"2021-08-23T17:06:53Z","snapshot_observed_at":"2026-08-12T07:28:34.615421Z","submitted_at":"2021-07-27T10:11:41Z","title":"End-to-End Spectro-Temporal Graph Attention Networks for Speaker Verification Anti-Spoofing and Speech Deepfake Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.12710","snapshot_observed_at":"2026-08-05T13:55:17.884576Z","title":"End-to-end spectro-temporal graph attention networks for speaker verification anti-spoofing and speech deepfake detection,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.884576Z"},"links":{"cited_paper":"/paper/2107.12710","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:d6974385ba4e0af62ac2daaace57aa6b8cecb4146a4e168aa9c65bd986cc7914","observation_id":"0cf6da84-779d-4589-8cf6-a1376b3101e1","resolution":{"observed_at":"2026-08-05T13:55:17.884576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:17.966404Z","title":"Speaker recognition from raw waveform with sincnet,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:17.966404Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:ce9cb31fda3764a5a907d1eddd9e1ae0fce25471a10f19e1ffaa9ab3f528990f","observation_id":"2870a2f9-4a44-4396-9947-e5fb20705d84","resolution":{"observed_at":"2026-08-05T13:55:17.966404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:22.466078Z","title":"Advanced rawnet2 with attention- based channel masking for synthetic speech detection,","venue":null,"work_id":"569092cb-caad-46c8-a6ba-4d89a60e1cb5","year":2023},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.060418Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:7288ea8897748a61fc12962558c94bbb91e19c86b11a5d60b6c532e526af5647","observation_id":"1430422c-6d30-4013-8b77-0a4f5d16045b","resolution":{"observed_at":"2026-08-05T13:55:22.606028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2002.12764","last_updated":"2020-08-06T04:53:37Z","snapshot_observed_at":"2026-08-11T06:28:46.409000Z","submitted_at":"2020-02-25T21:38:24Z","title":"Towards Learning a Universal Non-Semantic Representation of Speech","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.12764","snapshot_observed_at":"2026-08-05T13:55:18.816682Z","title":"T owards learning a universal non-semantic representation of speech,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.816682Z"},"links":{"cited_paper":"/paper/2002.12764","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:e5560a432bcc5ccb10543f4acd054734c0d33df7ccdcaec3d3b1d3e32a794c1c","observation_id":"0944ebae-b4b0-447c-9692-79e4e917056f","resolution":{"observed_at":"2026-08-05T13:55:18.816682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.925690Z","title":"A robust audio deepfake detection system via multi-view feature,","venue":null,"work_id":"423f9061-c57f-49d6-9c83-f70695f3d19b","year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.190662Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:4138a2e6d4784b07eb0af764e62bc1ccf930e5117d1e75ba60b1fd0a27399061","observation_id":"63d176cf-b1fb-445c-a424-4952d9f97d72","resolution":{"observed_at":"2026-08-05T13:55:22.052308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-08-05T13:55:18.277810Z","title":"Xls-r: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.277810Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:0c3f01f8680b5b9af52ae73e5efec11d7528a43f389ba430e031c5aa58c9279c","observation_id":"b1e5f998-c1c8-4785-91e5-dd8b66b957c6","resolution":{"observed_at":"2026-08-05T13:55:18.277810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.690397Z","title":"W avLM: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":"33d74146-88ab-4db9-96cc-29efa0c7861a","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.361581Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:f204a38ec28c72efda99a777c3c5608022ab58df11124f7e257af86a02e49dd2","observation_id":"624e3ae5-2f4a-48f8-923b-1d257798571b","resolution":{"observed_at":"2026-08-05T13:55:21.827909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.450193Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":"f99ea37b-8376-4827-b9d6-4e23b5b01ec3","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.467070Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:5ebd0cbce3470c77d66da6c4934387e20b82c66ec2405c8882bd155e127923ac","observation_id":"a4f6b350-65ae-4041-86d2-51f59f3be15c","resolution":{"observed_at":"2026-08-05T13:55:21.602155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.226479Z","title":"Exploring generalization to unseen audio data for spoof- ing: Insights from ssl models,","venue":null,"work_id":"c5d82e27-5268-4d15-b03d-c62648cf1002","year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.555336Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:328ff6d82e656c7ff72080152bc4ee9e784bef39b945f19b202228e3633c3e0a","observation_id":"c79c5911-4407-4dbc-958a-1adc6118ac81","resolution":{"observed_at":"2026-08-05T13:55:21.326332Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.07143","last_updated":"2020-08-10T13:50:24Z","snapshot_observed_at":"2026-08-07T15:21:18.262728Z","submitted_at":"2020-05-14T17:02:15Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.07143","snapshot_observed_at":"2026-08-05T13:55:18.642650Z","title":"Ecapa-tdnn: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.642650Z"},"links":{"cited_paper":"/paper/2005.07143","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:803a71156d84071e5d7edd829cf811260c40a8fe6d87b17a9b64353fea31eccd","observation_id":"99ef9ae6-dc3b-4e3b-8422-a1ff46782b01","resolution":{"observed_at":"2026-08-05T13:55:18.642650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09512","last_updated":"2026-05-18T13:28:38Z","snapshot_observed_at":"2026-07-06T17:17:03.447619Z","submitted_at":"2024-01-17T15:09:02Z","title":"MLAAD: The Multi-Language Audio Anti-Spoofing Dataset","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.09512","snapshot_observed_at":"2026-08-05T13:55:18.739632Z","title":"MLAAD: The multi-language audio anti-spoofing dataset,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.739632Z"},"links":{"cited_paper":"/paper/2401.09512","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:7474166d4f09e170b72b63a4ba74cea041921acb94a8a6c9faf9b6877f9c83e2","observation_id":"e25cfc55-f110-4080-aefb-f4ff357bbc0a","resolution":{"observed_at":"2026-08-05T13:55:18.739632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.661889Z","title":"Audio deepfake detection with self-supervised xls-r and sls classifier,","venue":null,"work_id":"2008256f-8c8f-4ae6-9196-074add1ed88b","year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.455717Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:bf5ff8a49f7514f35bd2f5bb7996f30addfd92f2b74dfc58696b8fda21c1b7e3","observation_id":"0c57f54a-865a-4372-b2bd-7ea874dddeec","resolution":{"observed_at":"2026-08-05T13:55:20.719919Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:24.871664Z","title":"This gives us insight that features extracted over a duration of average syllable length are more beneficial for spoofing detection tasks","venue":null,"work_id":"094f8bf3-3a81-4006-ae9d-81c2b8d5f988","year":null},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:16.685798Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:163294d51925e5dc4cc1479c92edad68cc8afefa19105c5cea05295301edf8f4","observation_id":"cf2c4358-cce5-4475-836e-1ce2b613bb92","resolution":{"observed_at":"2026-08-05T13:55:24.939678Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.00236","last_updated":"2022-03-20T21:13:37Z","snapshot_observed_at":"2026-08-02T20:59:20.503566Z","submitted_at":"2022-03-01T05:22:57Z","title":"TRILLsson: Distilled Universal Paralinguistic Speech Representations","version":2},"cited_work":{"arxiv_id":"2203.00236","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.00236","snapshot_observed_at":"2026-08-05T13:55:19.919132Z","title":"TRILLsson: Distilled Universal Paralinguistic Speech Representations","venue":"eess.AS","work_id":"2667345a-4afe-4d53-bbe0-1b2bfb474257","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.869944Z"},"links":{"cited_paper":"/paper/2203.00236","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:f80f2c48d764e7c9c1d87d61a746c3fd24b3870cec98ce80b943906593743681","observation_id":"572231c0-dbee-430e-8b47-9a111b99a5f7","resolution":{"observed_at":"2026-08-05T13:55:19.972447Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.087789Z","title":"Self- normalizing neural networks,","venue":null,"work_id":"4d39ea79-090f-4813-8d93-a3a093d29213","year":2017},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:18.957534Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:ea54e3d422a64b25abdd175c8b574ed8f58b0ccebd1138a7facf214aeb6b538c","observation_id":"fc7f36cd-3d29-4d44-ab0b-1efb3dab4bcb","resolution":{"observed_at":"2026-08-05T13:55:21.146141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:21.014886Z","title":"CSTR VCTK Corpus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92),","venue":null,"work_id":"9c1b14b9-bb98-481d-a4ec-d06f8589559e","year":2019},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.066896Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:eb1f1f8d18cb1041bfd69b3360a7b8af471bb0c3a34dfce39c4a97ef439576f5","observation_id":"c6e03aa8-a717-4527-a7f9-b45650198a11","resolution":{"observed_at":"2026-08-05T13:55:21.034346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.918794Z","title":"ASVspoof 2021: T owards spoofed and deepfake speech detection in the wild,","venue":null,"work_id":"f6a3e4fc-2e76-4356-a23d-cba450719c37","year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.131502Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:4767549c66e15ee94dfec11596203ccf8dbe62a6eb9885823628ad99a8332454","observation_id":"bf1a2af7-3b66-4177-aa54-39c323f20fd6","resolution":{"observed_at":"2026-08-05T13:55:20.963937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.07725","last_updated":"2022-02-04T13:25:23Z","snapshot_observed_at":"2026-08-03T12:01:20.906827Z","submitted_at":"2021-11-15T12:52:50Z","title":"Investigating self-supervised front ends for speech spoofing countermeasures","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.07725","snapshot_observed_at":"2026-08-05T13:55:19.205724Z","title":"Investigating self-supervised front ends for speech spoofing countermeasures,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.205724Z"},"links":{"cited_paper":"/paper/2111.07725","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:454cf10fc1c3500d271703c6ce371a3af252731088161b8e3b195a1df257eeea","observation_id":"d88ac0a8-0e34-4277-8ab3-641c88959888","resolution":{"observed_at":"2026-08-05T13:55:19.205724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.776555Z","title":"Lever- aging positional-related local-global dependency for synthetic speech detection,","venue":null,"work_id":"26ff4d59-dc22-44a3-8f85-b6c908e7bd47","year":2023},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.269691Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:41206e3ea8c53c7d2a2b294dcb0b4ebfc290367d4b387400fb3b943781490247","observation_id":"4f3ef205-f176-42e6-bb65-116e874b1ad4","resolution":{"observed_at":"2026-08-05T13:55:20.848923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13495","last_updated":"2024-10-31T09:11:37Z","snapshot_observed_at":"2026-08-12T04:55:35.429836Z","submitted_at":"2024-06-19T12:35:02Z","title":"DF40: Toward Next-Generation Deepfake Detection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13495","snapshot_observed_at":"2026-08-05T13:55:19.374122Z","title":"Df40: T oward next-generation deepfake detection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.374122Z"},"links":{"cited_paper":"/paper/2406.13495","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:0f559fa9443caef65673616cb9838d1558225c1143e69d370daa1cb1ddde7638","observation_id":"4f7c1ed3-d8e2-491a-bf7a-1384124f87e6","resolution":{"observed_at":"2026-08-05T13:55:19.374122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.03812","last_updated":"2022-03-10T01:43:15Z","snapshot_observed_at":"2026-07-06T12:45:21.570121Z","submitted_at":"2022-03-08T02:22:28Z","title":"SpeechFormer: A Hierarchical Efficient Framework Incorporating the Characteristics of Speech","version":2},"cited_work":{"arxiv_id":"2203.03812","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.03812","snapshot_observed_at":"2026-08-05T13:55:19.763491Z","title":"SpeechFormer: A Hierarchical Efficient Framework Incorporating the Characteristics of Speech","venue":"cs.SD","work_id":"68bbb993-507f-44cd-bd1e-a34239a0038a","year":2022},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.542747Z"},"links":{"cited_paper":"/paper/2203.03812","citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:5cfee50821b09a2ed0893e821c408078b76e21516b523cf3e934b2abd05e6404","observation_id":"7a2a2976-620d-49fe-b59d-102ed3645829","resolution":{"observed_at":"2026-08-05T13:55:19.811280Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:55:20.548738Z","title":"Syllable-level duration determination","venue":null,"work_id":"1bdd1dd2-7e1e-4417-9835-53b43613523f","year":1989},"citing_paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-05T13:55:19.657537Z"},"links":{"citing_paper":"/paper/2509.00186"},"observation_digest":"sha256:4b2ac02aa1ba8de84636aa8b079ba18cb2be7b07a8c9f3e1f4f4a56dc1da45d4","observation_id":"451cc3be-3a1c-4e1d-ad45-49537757995c","resolution":{"observed_at":"2026-08-05T13:55:20.611212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.00186","last_updated":"2025-08-29T18:37:57Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-07T21:38:02.692073Z","submitted_at":"2025-08-29T18:37:57Z","title":"Generalizable Audio Spoofing Detection using Non-Semantic Representations"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":11,"verified_exact":3,"verified_fuzzy":25},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 12 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 2 inbound Pith citation observations for arXiv:2509.00186."}