{"as_of":"2026-08-17T02:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a261ae745656454736cf336a429d54959d4b956938d1a23762e3b7ac5472c5f6","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T15:53:11.146098Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.13811/citation-record","integrity":"/paper/2411.13811/integrity","json":"/paper/2411.13811/citation-record.json","paper":"/paper/2411.13811"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.477923Z","title":"Deep clustering: Discriminative embeddings for segmentation and separation,","venue":null,"work_id":"c5b4708e-1f8b-486d-888a-6802224ddea8","year":2016},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.024260Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:18076cb5360669dfd49866d1d1871db58effacbc9919766119cf31b855080f27","observation_id":"f7a604b1-96b1-4575-8948-c2f63433021d","resolution":{"observed_at":"2026-08-12T15:53:11.482346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2010.12766","last_updated":"2020-10-24T03:57:19Z","snapshot_observed_at":"2026-08-16T19:10:59.181582Z","submitted_at":"2020-10-24T03:57:19Z","title":"X-TaSNet: Robust and Accurate Time-Domain Speaker Extraction Network","version":1},"cited_work":{"arxiv_id":"2010.12766","doi":null,"metadata_source":"pith","pith_arxiv_id":"2010.12766","snapshot_observed_at":"2026-08-12T15:53:11.290499Z","title":"X-TaSNet: Robust and Accurate Time-Domain Speaker Extraction Network","venue":"eess.AS","work_id":"dd65dcc7-dbb0-458a-b285-718623cb2b63","year":2020},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.029868Z"},"links":{"cited_paper":"/paper/2010.12766","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:35a92c8a7804e5abe53480a9a0c3286b971f6eceaed3cd4d7c7b2cb70996bb53","observation_id":"04028720-58e2-4d16-9c4d-adc632d03d8a","resolution":{"observed_at":"2026-08-12T15:53:11.297492Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.035688Z","title":"Conv-tasnet: Surpassing ideal time– frequency magnitude masking for speech separation,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.035688Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:7285991b014aef9868b8938e0d5582ac47ec218055a4526f2db8481bb717e466","observation_id":"e79aa124-4c5b-4036-a234-e0a6391e6eb4","resolution":{"observed_at":"2026-08-12T15:53:11.035688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.459287Z","title":"X-sepformer: End-to-end speaker extraction network with explicit optimization on speaker confusion,","venue":null,"work_id":"791f6148-6321-457a-84e2-dd8f230dbadb","year":2023},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.040021Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:970cdeb35e034def0937e1bca7c344efb6eec658613845d6664f53f6f6ad7054","observation_id":"22ae94b1-0ec2-49a6-82f5-d78522e1e266","resolution":{"observed_at":"2026-08-12T15:53:11.463604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.044480Z","title":"Attention is all you need in speech separation,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.044480Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:38f0f54221d9d7927eb198e556fe6726a5bd8e04e1099741eddf3bde02ad753d","observation_id":"6ab5cef4-d449-410e-9d3a-646586da20d4","resolution":{"observed_at":"2026-08-12T15:53:11.044480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.441688Z","title":"X-tf-gridnet: A time–frequency domain target speaker extraction network with adaptive speaker embedding fusion,","venue":null,"work_id":"58f4fa29-5610-4bb2-acb0-c37ca4cee67f","year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.048595Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:e67b6f9887b0e616f3619f662c4386687f4503121927275fa365f9ebfca15089","observation_id":"b5b27d92-39b9-4bb9-962d-0d7982f12f4c","resolution":{"observed_at":"2026-08-12T15:53:11.445701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.430512Z","title":"Tf-gridnet: Making time-frequency domain models great again for monaural speaker separation,","venue":null,"work_id":"4b887623-5369-4d21-a723-093f1a521591","year":2023},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.053272Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:aa4720799a617bc7afe569dd0b81ef4d6546be10614981072f7a571ff234e2bb","observation_id":"71ff4977-d239-43e8-9f4d-f8fb153437bc","resolution":{"observed_at":"2026-08-12T15:53:11.434340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.419189Z","title":"Single channel target speaker extraction and recognition with speaker beam,","venue":null,"work_id":"3c8228bc-2726-4c4c-92c9-ef98df24ad4e","year":2018},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.057050Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:66f9985bb163a76c0228157d1b311609369edef48c6bcbd6b1eaf3a1ddbfe078","observation_id":"b993366f-2619-4d54-b3dc-91e0af823588","resolution":{"observed_at":"2026-08-12T15:53:11.423397Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04826","last_updated":"2019-06-19T17:10:51Z","snapshot_observed_at":"2026-08-14T18:16:27.334746Z","submitted_at":"2018-10-11T02:57:14Z","title":"VoiceFilter: Targeted Voice Separation by Speaker-Conditioned Spectrogram Masking","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04826","snapshot_observed_at":"2026-08-12T15:53:11.061029Z","title":"V oicefilter: Targeted voice separation by speaker-conditioned spectrogram masking,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.061029Z"},"links":{"cited_paper":"/paper/1810.04826","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:095f0b03ef843b5a18660f4ecbbd50e12b6faf57f59fde79b3f0e88c1f731e88","observation_id":"8841318e-8b8e-458b-876c-4e54099e768f","resolution":{"observed_at":"2026-08-12T15:53:11.061029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2202.00733","last_updated":"2023-09-15T06:15:22Z","snapshot_observed_at":"2026-08-16T17:24:17.008198Z","submitted_at":"2022-02-01T20:10:23Z","title":"New Insights on Target Speaker Extraction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2202.00733","snapshot_observed_at":"2026-08-12T15:53:11.066619Z","title":"New insights on target speaker extraction,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.066619Z"},"links":{"cited_paper":"/paper/2202.00733","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:a54435554391f8af39961c7d8e82742bd5437862885bc9a23f3cf94e08109d5f","observation_id":"27de0348-9ded-4fb5-97f8-4aec62bab1ba","resolution":{"observed_at":"2026-08-12T15:53:11.066619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.071241Z","title":"Wavesplit: End-to-end speech separation by speaker clustering,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.071241Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:aac6627cc47dcfd11eb5ebc94b4cb31ab9c99361f9ddd55e67ce53c9a9c70c50","observation_id":"c6689993-6469-480e-9fe0-dcd336a4d983","resolution":{"observed_at":"2026-08-12T15:53:11.071241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.399948Z","title":"Attention-based scaling adaptation for target speech extraction,","venue":null,"work_id":"205344cf-fb42-45b3-a054-7fcb7f97c90c","year":2021},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.075799Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:9417668ea666ef0c14e9aa5c0bfda8d71e2d1317ea799e7ae6f64862a30a5b8d","observation_id":"4ac044fc-519e-4126-9605-4f399e024254","resolution":{"observed_at":"2026-08-12T15:53:11.404027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.386784Z","title":"Target speaker extraction by directly exploiting contextual information in the time-frequency domain,","venue":null,"work_id":"88e35314-6697-48f4-8e4e-03bc11c14025","year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.083679Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:4e0f18505b84644f6436889d95de7a4b56c9c2bc216c132498ec1f7b7b33095b","observation_id":"a3cee24f-c101-48da-b1e6-300af3a912b7","resolution":{"observed_at":"2026-08-12T15:53:11.392409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03411","last_updated":"2024-03-06T02:39:21Z","snapshot_observed_at":"2026-08-16T14:12:27.394190Z","submitted_at":"2024-03-06T02:39:21Z","title":"CrossNet: Leveraging Global, Cross-Band, Narrow-Band, and Positional Encoding for Single- and Multi-Channel Speaker Separation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.03411","snapshot_observed_at":"2026-08-12T15:53:11.087244Z","title":"Crossnet: Leveraging global, cross- band, narrow-band, and positional encoding for single-and multi-channel speaker separation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.087244Z"},"links":{"cited_paper":"/paper/2403.03411","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:9dec7accae0443bcc813c4b1e73ef7d538c3f365cba3531f0b3332b735a37dac","observation_id":"2e8fd2de-7ed2-433d-9b4c-5ca83e305eee","resolution":{"observed_at":"2026-08-12T15:53:11.087244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.08100","last_updated":"2020-05-16T20:56:25Z","snapshot_observed_at":"2026-08-15T19:06:16.181826Z","submitted_at":"2020-05-16T20:56:25Z","title":"Conformer: Convolution-augmented Transformer for Speech Recognition","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.08100","snapshot_observed_at":"2026-08-12T15:53:11.092180Z","title":"Conformer: Convolution-augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.092180Z"},"links":{"cited_paper":"/paper/2005.08100","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:3b325dc7caf86b613425421d60a70270f94b77e86e8125bff083b39a00467409","observation_id":"4f747a62-0642-4d58-9448-abbc1ee71cb4","resolution":{"observed_at":"2026-08-12T15:53:11.092180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.095809Z","title":"Sdr–half-baked or well done?","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.095809Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:78c371ffd53a94ba1ecfd0d40d42d96f95727f1835c5af74d7ebf43a051cedec","observation_id":"ce92acf1-fd2f-4e0d-ab46-db00a288aadd","resolution":{"observed_at":"2026-08-12T15:53:11.095809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2005.04686","last_updated":"2020-08-18T03:03:01Z","snapshot_observed_at":"2026-08-12T03:39:08.946505Z","submitted_at":"2020-05-10T15:00:07Z","title":"SpEx+: A Complete Time Domain Speaker Extraction Network","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.04686","snapshot_observed_at":"2026-08-12T15:53:11.099817Z","title":"Spex+: A complete time domain speaker extraction network,","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.099817Z"},"links":{"cited_paper":"/paper/2005.04686","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:fb8c385758fb5d48ab17ca9edd3c0eeff54661b4392b80d95d377bda4ed784a4","observation_id":"5a0d7351-89b4-4407-8c88-f4994ad21691","resolution":{"observed_at":"2026-08-12T15:53:11.099817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.364489Z","title":"Whamr!: Noisy and reverberant single-channel speech separation,","venue":null,"work_id":"22dd03e4-6de1-4d38-95dd-d6f56cf7d84f","year":2020},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.106726Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:0b9a0e5eab12f21893b8a877660bcb7e7d893f8e9b353350be331e1416d1dc87","observation_id":"1b66e01e-0d20-4204-9868-480abdfaf3b2","resolution":{"observed_at":"2026-08-12T15:53:11.369502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1907.01160","last_updated":"2019-07-02T04:27:55Z","snapshot_observed_at":"2026-07-06T08:04:17.909965Z","submitted_at":"2019-07-02T04:27:55Z","title":"WHAM!: Extending Speech Separation to Noisy Environments","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1907.01160","snapshot_observed_at":"2026-08-12T15:53:11.110936Z","title":"Wham!: Extending speech separation to noisy environments,","venue":null,"work_id":null,"year":1907},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.110936Z"},"links":{"cited_paper":"/paper/1907.01160","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:ff55be9cd9c751381d2c613e8efe8e109a63a853f014eca4f70d56c573cb79e4","observation_id":"12820096-87ab-49d4-b27a-b51b73842a0e","resolution":{"observed_at":"2026-08-12T15:53:11.110936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.350577Z","title":"Time-domain speaker extraction network,","venue":null,"work_id":"319f3e9c-8d8e-426d-8c11-da011423a974","year":2019},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.119063Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:d9c97975919761af22312496a35f0fd877def30342ae052782a7579d3aeecc9c","observation_id":"666982a8-6f6a-487e-beb6-e6c9e047c42f","resolution":{"observed_at":"2026-08-12T15:53:11.354880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.337058Z","title":"Spex: Multi-scale time domain speaker extraction network,","venue":null,"work_id":"f25dfe5c-a22c-4e7f-899e-e3af7387f730","year":2020},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.124767Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:a37908d593d7e8c0d7ed90e276948ee638971d79884cf316a62e8fd6d93201e1","observation_id":"1a39e866-ee0b-415b-80e4-a2c74f7f8a0f","resolution":{"observed_at":"2026-08-12T15:53:11.341213Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.325156Z","title":"Sef-net: Speaker embedding free target speaker extraction network,","venue":null,"work_id":"82d8dc39-adba-4f6d-92df-71469b30a294","year":2023},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.128516Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:4a9a8c4f623e0574f36928e2cc6cf8dc34199a6e4e082ea5c291bdad7a3e0a93","observation_id":"e855fa65-3bdb-4a7a-b8a3-aea102bad867","resolution":{"observed_at":"2026-08-12T15:53:11.329226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.134385Z","title":"Spatialnet: Extensively learning spatial information for multichannel joint speech separation, denoising and dereverberation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.134385Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:bd1ba5b25e02083b690bf866d87508ca2bac7a795aef1a10ad9ca8d897d9edc0","observation_id":"610054a0-a670-464d-bde2-d4c5d665fedc","resolution":{"observed_at":"2026-08-12T15:53:11.134385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1711.05101","last_updated":"2019-01-04T21:01:49Z","snapshot_observed_at":"2026-08-14T20:13:52.872565Z","submitted_at":"2017-11-14T14:24:06Z","title":"Decoupled Weight Decay Regularization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.05101","snapshot_observed_at":"2026-08-12T15:53:11.139423Z","title":"Decoupled weight decay regularization,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.139423Z"},"links":{"cited_paper":"/paper/1711.05101","citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:fa9952d69371c516bc277b8bb534ae5fd46eccffc98d469e9d2b0d8cf0e961d4","observation_id":"e8685796-2d3d-47df-bf46-211787792bf9","resolution":{"observed_at":"2026-08-12T15:53:11.139423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T15:53:11.305298Z","title":"Performance measurement in blind audio source separation,","venue":null,"work_id":"af112a6a-e922-425d-b3b3-9403f7b5bf3b","year":2006},"citing_paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T15:53:11.146098Z"},"links":{"citing_paper":"/paper/2411.13811"},"observation_digest":"sha256:815913f920eadbb84f646f5adf9cc21176696e1f2bebae110e454651d96b07a6","observation_id":"286d71a3-0744-4ae6-95ab-d92c2b8aa724","resolution":{"observed_at":"2026-08-12T15:53:11.308970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.13811","last_updated":"2024-11-25T00:21:53Z","latest_version":2,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-14T00:03:13.464897Z","submitted_at":"2024-11-21T03:21:42Z","title":"X-CrossNet: A complex spectral mapping approach to target speaker extraction with cross attention speaker embedding fusion"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":1,"verified_fuzzy":12},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2411.13811."}