{"as_of":"2026-08-10T14:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:03f65ee4f6ef61af5a160b6ab804e21f2856ee8448afd0ac5dbd775583f9242e","coverage":[{"denominator":30,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:33:14.007487Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:33:13.912562Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T14:33:14.053965Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"cited_work":{"arxiv_id":"2505.18498","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.18498","snapshot_observed_at":"2026-08-07T14:33:14.053965Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","venue":"cs.SD","work_id":"82583725-9e55-44d6-b1f4-969cbfb833f7","year":2025},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.912562Z"},"links":{"cited_paper":"/paper/2505.18498","citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:719bb4bbbbfb7d1ac07c090b63ff88145b472797e3bff0d1d7e8cb13dced1366","observation_id":"1c53738c-ad4d-4cb8-b7b5-c8af21ec7dfb","resolution":{"observed_at":"2026-08-07T14:33:14.058991Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.18498/citation-record","integrity":"/paper/2505.18498/integrity","json":"/paper/2505.18498/citation-record.json","paper":"/paper/2505.18498"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.354093Z","title":"Currently, SV systems that use low- dimensional speaker representations extracted from deep learning- based speaker encoders have become the dominant approach in this field","venue":null,"work_id":"42cf79d2-bf9b-4d25-8303-b8aaa2037cf1","year":null},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.908650Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:ca7b7c828122b37f19415f0fce58f203c7f9d6a6d43b1f6f76d96733895f9b0f","observation_id":"d272e83d-db89-464e-a96d-f5c90b0553ec","resolution":{"observed_at":"2026-08-07T14:33:14.358571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"cited_work":{"arxiv_id":"2505.18498","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.18498","snapshot_observed_at":"2026-08-07T14:33:14.053965Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","venue":"cs.SD","work_id":"82583725-9e55-44d6-b1f4-969cbfb833f7","year":2025},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.912562Z"},"links":{"cited_paper":"/paper/2505.18498","citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:719bb4bbbbfb7d1ac07c090b63ff88145b472797e3bff0d1d7e8cb13dced1366","observation_id":"1c53738c-ad4d-4cb8-b7b5-c8af21ec7dfb","resolution":{"observed_at":"2026-08-07T14:33:14.058991Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.342242Z","title":"Datasets Our models are first pre-trained on the V oxCeleb [22] and then fine- tuned on the Dusha [18] dataset to evaluate the performance of SV","venue":null,"work_id":"6581353d-cea3-4740-8bdc-aeb7b3dec9b6","year":null},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.916593Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:aa2d4dd818fd0040b93f2a6a9ece30bec16e3f0d6bb8dadce5d1af7dc62b386a","observation_id":"ef6aa4ac-7b3b-43de-9116-967227f13384","resolution":{"observed_at":"2026-08-07T14:33:14.346234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.331050Z","title":"Performance of the baseline system The performance of the pre-trained model on V oxCeleb1 is presented in Table 3, with the results of the ECAPA model referenced from [3]","venue":null,"work_id":"5d8d3076-62a1-46a0-b4fd-30896e2cf643","year":null},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.920331Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:36ac9a9bc1232db02ebd793ded19e89c2ee775893ea6380f42c51787353765f4","observation_id":"e4b5f3c6-29e9-4647-b9dd-79d75c8cae9a","resolution":{"observed_at":"2026-08-07T14:33:14.335678Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.319389Z","title":"We first verified that emotional utterances degrade SV performance, with cross-emotion test trails performed worse than same-emotion test trails","venue":null,"work_id":"770f5e7a-8943-4595-a661-ecb848cbcef6","year":null},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.923606Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:86968ac8500c7c67d4d0c0f0d6f6f397e06487b5342eda1d0184e23d411e54ab","observation_id":"51d46830-4f89-40c2-9dd7-eb89b07fa94e","resolution":{"observed_at":"2026-08-07T14:33:14.323742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.308718Z","title":"X-vectors: Robust dnn em- beddings for speaker recognition,","venue":null,"work_id":"ecf8e581-f27b-43da-98b5-98919efa07b6","year":2018},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.926783Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:fefc77661da8e642c6ed68865e056d12325e1022859341fd8cbfb62ce745a008","observation_id":"e6b4887c-d265-4784-9cd3-313a6928f3c6","resolution":{"observed_at":"2026-08-07T14:33:14.312011Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.12592","last_updated":"2019-10-16T11:27:27Z","snapshot_observed_at":"2026-08-09T06:19:25.884792Z","submitted_at":"2019-10-16T11:27:27Z","title":"BUT System Description to VoxCeleb Speaker Recognition Challenge 2019","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.12592","snapshot_observed_at":"2026-08-07T14:33:13.929697Z","title":"But system description to vox- celeb speaker recognition challenge 2019,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.929697Z"},"links":{"cited_paper":"/paper/1910.12592","citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:80116ac822a56debb30e64b54ae08a0f71a027e1d8b1d3dcbb15cb74e74cef18","observation_id":"4c2a3cbe-4731-48f2-836e-011dc408225f","resolution":{"observed_at":"2026-08-07T14:33:13.929697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.296520Z","title":"ECAPA-TDNN: Emphasized Channel Attention, Propagation and Aggregation in TDNN Based Speaker Verification,","venue":null,"work_id":"d2caf5a3-034f-464c-806b-0ed8fa61a536","year":2020},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.933411Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:53987c536058f7ff7825693041e925941dacf2a98003724de3c6a11366825b59","observation_id":"acb2c15c-c308-4464-85c9-2b6052663fdb","resolution":{"observed_at":"2026-08-07T14:33:14.300839Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.286451Z","title":"MFA-Conformer: Multi-scale Feature Aggregation Con- former for Automatic Speaker Verification,","venue":null,"work_id":"97ff45f2-377b-44b1-9371-1a1da28b9d24","year":2022},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.937296Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:24ad0f52c6771404d92f5d7f18405b9f0302013651c3da8e58702ef6cc5c3eec","observation_id":"451a8d6f-545d-4d5f-966b-bf4bb031e276","resolution":{"observed_at":"2026-08-07T14:33:14.289827Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.274444Z","title":"Margin matters: Towards more discriminative deep neural network embeddings for speaker recognition,","venue":null,"work_id":"e2929f15-ac6d-4dc3-987d-fd539ca85aab","year":2019},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.940222Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:12cef638f567e2d812215b3c95f0408e80ff85b7e999c6b51023eb6fd33eeeed","observation_id":"662e5a3b-e0d7-424e-bc6b-256c079af3e4","resolution":{"observed_at":"2026-08-07T14:33:14.279108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.264016Z","title":"In Defence of Metric Learn- ing for Speaker Recognition,","venue":null,"work_id":"ce86fca6-76f6-470e-984a-f6b338f6b676","year":2020},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.944520Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:225137ac3c957d511b8a0a955ca4b0f1369c354e6997bc2b59704a7a575a32e0","observation_id":"447260bd-c7cd-4507-916b-ec3ed4f55f6a","resolution":{"observed_at":"2026-08-07T14:33:14.267645Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.252591Z","title":"Multi-query multi-head attention pooling and inter-topk penalty for speaker verification,","venue":null,"work_id":"40b709b6-5299-428f-b6cd-a308e600323c","year":2022},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.948506Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:be64ef1c06ebc6a8cfdc319fc8e9d523c30b71f55e98803735bc565264991df5","observation_id":"49ed5cc9-3b10-4dea-a8fb-32e56cc92aeb","resolution":{"observed_at":"2026-08-07T14:33:14.256580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.239006Z","title":"Explor- ing binary classification loss for speaker verification,","venue":null,"work_id":"b327695b-4256-4d39-b1a1-5569f9cb0472","year":2023},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.952200Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:52990895cc22b77bcab006b10ecb8691b5d5ee8a40c6574f0a980f369bc3a681","observation_id":"ca578001-9df4-4d34-95a1-ec41e6d25e45","resolution":{"observed_at":"2026-08-07T14:33:14.242783Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.227162Z","title":"Nplda: A deep neural plda model for speaker verification,","venue":null,"work_id":"293cc4ce-e647-421d-b6e7-a15da59895c3","year":2020},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.956178Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:46819a68ffeadda2a78e5d7c09bd29a929e6a5773bf642dcfe05556c1936a6a3","observation_id":"6cf5c2e2-72ec-481d-a0c4-de75ad522a61","resolution":{"observed_at":"2026-08-07T14:33:14.231624Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.215449Z","title":"Scoring of Large-Margin Embeddings for Speaker Verification: Cosine or PLDA?,","venue":null,"work_id":"127ac729-6ad5-4ebe-be8c-c0435f425bf1","year":2022},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.959639Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:facf8bed39d4bcf07d7cb5e058c7bd568975f4dfaf93d2bba59296b4ac46ec84","observation_id":"ca1df279-d672-42a6-8fad-01f644036535","resolution":{"observed_at":"2026-08-07T14:33:14.219413Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.203449Z","title":"Attention back-end for automatic speaker verification with multiple enrollment utterances,","venue":null,"work_id":"59adc248-9940-4c65-af13-afe0afe79432","year":2022},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.962936Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:4b7eeb00ed409a5cd33abd9fd8ac8789e711dcadc483bbf77e65ed837c148f3b","observation_id":"cc0dba39-3f10-4b23-a105-ad9b91c58656","resolution":{"observed_at":"2026-08-07T14:33:14.207711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.193524Z","title":"Prob- abilistic Spherical Discriminant Analysis: An Alternative to PLDA for length-normalized embeddings,","venue":null,"work_id":"933d1661-52f2-4f68-b872-26ee73df59d2","year":2022},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.966314Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:87b3bcc2448e7f123bd5714c882e533996c5631f59058404b400963f2285549b","observation_id":"8fb1c116-b121-4d24-ac15-c9e94d28bc45","resolution":{"observed_at":"2026-08-07T14:33:14.196784Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.182179Z","title":"A study of speaker verification performance with expressive speech,","venue":null,"work_id":"2cb9c01a-fa1c-46dd-963c-4d1842c0c442","year":2017},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.968866Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:926cfb12d8940deb68e6e45a5603a761b786127b4360bd18a3f2ab4cb559f9d3","observation_id":"7a8bbea9-f537-430f-9007-e5b796bfa35a","resolution":{"observed_at":"2026-08-07T14:33:14.185802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.170938Z","title":"x-vectors meet emotions: A study on dependencies between emotion and speaker recognition,","venue":null,"work_id":"5163525b-df0b-45a7-9567-14c8fc332c0b","year":2020},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.971447Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:6527e8ffa6c0071d786fe42e322e74cbe60eff84586a03e7cddbc61c2756e460","observation_id":"975d0073-5454-4f20-b597-0ea549de3dd3","resolution":{"observed_at":"2026-08-07T14:33:14.175812Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.158744Z","title":"Emotion attribute projection for speaker recognition on emo- tional speech,","venue":null,"work_id":"814a2150-d529-40d4-a6f4-1eb8ecd78d5d","year":2007},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.974634Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:ccce5a7673cf718cb51a50babe86fc397eb7e2bdf8dd8acf624ce1f19e1ce396","observation_id":"355036ba-0493-4a31-888b-0724812840b3","resolution":{"observed_at":"2026-08-07T14:33:14.163436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.147250Z","title":"Segment- Level Effects of Gender, Nationality and Emotion Informa- tion on Text-Independent Speaker Verification,","venue":null,"work_id":"2822f2ab-59a4-44e1-9684-97b2f8c42a90","year":2020},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.977992Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:636e74cea9db331b36c44e390dd538913917e3d67de9a5d526bf46c87603797b","observation_id":"cf971125-39da-41ee-9c76-acbc9fee0ae7","resolution":{"observed_at":"2026-08-07T14:33:14.151537Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.135718Z","title":"Instance-based Temporal Normalization for Speaker Verifi- cation,","venue":null,"work_id":"1000702b-3b7c-4a05-8700-696781972eaf","year":2023},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.981024Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:47a60e3244e9db1e407ae9bec698f578686b389952f7bb1967464b83264ea031","observation_id":"0a3ed5c0-f612-479b-98ef-9108cdf3f6e5","resolution":{"observed_at":"2026-08-07T14:33:14.139636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.124496Z","title":"Hy- brid Dataset for Speech Emotion Recognition in Russian Lan- guage,","venue":null,"work_id":"9f2a3b81-2a0e-443b-a07b-ce5f29e8121f","year":2023},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.984661Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:a1ed1f5d0af0dd9a9eee8a5c8f76eac8c28b240eec08d57d2bc6aefb71db263c","observation_id":"37549ff7-4e44-4184-a0a5-a54e562cd1f2","resolution":{"observed_at":"2026-08-07T14:33:14.128075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.113127Z","title":"Copypaste: An augmentation method for speech emotion recognition,","venue":null,"work_id":"98c77db3-b435-4124-a2f2-443eaf9081f0","year":2021},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.988353Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:261841f2331cfdbd0c5f3eeaddc68aafd2477bb64e86a4858e06cc91bc6eeef6","observation_id":"b7d2bb8d-0695-40f2-81fd-39aa33242bf7","resolution":{"observed_at":"2026-08-07T14:33:14.116391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:13.991313Z","title":"Arcface: Additive angular margin loss for deep face recogni- tion,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.991313Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:bb572ff34bd7e471bc288007e30916b4e855906d7384ea5b4c00ebbbb3e8c85e","observation_id":"992077d4-9643-4452-af41-65fa8401c116","resolution":{"observed_at":"2026-08-07T14:33:13.991313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.097338Z","title":"Sur- vey on speech emotion recognition: Features, classification schemes, and databases,","venue":null,"work_id":"4b483ce7-a47a-4a16-a9a8-7974d4b161ee","year":2011},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.993871Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:2ef8f5d6778732dfe766653e35587308772451d17831b4d661f1a88b7dd80123","observation_id":"32601c19-731e-4846-9047-ba2e1f0537ab","resolution":{"observed_at":"2026-08-07T14:33:14.100109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.087639Z","title":"V oxceleb: Large-scale speaker verification in the wild,","venue":null,"work_id":"9cbc08f7-d21e-4332-91e5-d1800cb3ce20","year":2020},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.996952Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:677a33dceaec6ea14a23e6dc17936f6d0fd02228c165380394812bfc8fe48d9a","observation_id":"e8787bbb-9e0f-4918-a1b3-ed414dc1f771","resolution":{"observed_at":"2026-08-07T14:33:14.090707Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.077559Z","title":"Re- visiting the statistics pooling layer in deep speaker embedding learning,","venue":null,"work_id":"9915836b-dc85-45c5-95cc-95c1fe968e2c","year":2021},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:13.999565Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:e0f1d012c002b31e193ae6e0302aec215f3b3094da663bdc48f3fdbdfcf85d4d","observation_id":"9fe6b536-cc38-4501-b8de-3d7e303c9a97","resolution":{"observed_at":"2026-08-07T14:33:14.080892Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1510.08484","last_updated":"2015-10-28T20:59:04Z","snapshot_observed_at":"2026-07-06T04:34:36.474437Z","submitted_at":"2015-10-28T20:59:04Z","title":"MUSAN: A Music, Speech, and Noise Corpus","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1510.08484","snapshot_observed_at":"2026-08-07T14:33:14.003406Z","title":"Mu- san: A music, speech, and noise corpus,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:14.003406Z"},"links":{"cited_paper":"/paper/1510.08484","citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:1740ce79075b20365be7a44f53ce830ca954396120e2a438e8697bbe4f8fdda8","observation_id":"da5662a4-e6e1-4ce4-a827-e6df1c00a14b","resolution":{"observed_at":"2026-08-07T14:33:14.003406Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:33:14.066849Z","title":"A study on data augmen- tation of reverberant speech for robust speech recognition,","venue":null,"work_id":"c1046fbf-ae96-4010-afde-709fb0956e3f","year":2017},"citing_paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:33:14.007487Z"},"links":{"citing_paper":"/paper/2505.18498"},"observation_digest":"sha256:142f2902b0b13dcc8178d0ad2adf668024043adbed512767dbccc76143cedbb1","observation_id":"647e1a2e-06b3-48ff-ac2e-916d31a40bdc","resolution":{"observed_at":"2026-08-07T14:33:14.070342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.18498","last_updated":"2025-05-24T04:31:12Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-09T06:20:02.235023Z","submitted_at":"2025-05-24T04:31:12Z","title":"Learning Emotion-Invariant Speaker Representations for Speaker Verification"},"reference_resolution":{"displayed":30,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":3,"verified_exact":0,"verified_fuzzy":25},"total_outbound_references":30},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 30 of 30 outbound references and 1 inbound Pith citation observation for arXiv:2505.18498."}