{"as_of":"2026-08-21T09:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:21814ee49069ca757228bbdc8a15224ee51ab05f08cafc169624a3f222d0056f","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T18:50:08.405900Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T18:50:08.261278Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-12T18:50:08.499324Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"cited_work":{"arxiv_id":"2411.11232","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.11232","snapshot_observed_at":"2026-08-12T18:50:08.499324Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","venue":"cs.SD","work_id":"8e46c1f2-bd7d-4d79-bf58-d19ed1dcf6cc","year":2024},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.261278Z"},"links":{"cited_paper":"/paper/2411.11232","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:5681136c9e893c64e16ea39de0f2fb0fe04f40d1f8534ceae7a9eb5dc88d016c","observation_id":"9029bd3b-988a-4b5d-a2ce-96aa0b219744","resolution":{"observed_at":"2026-08-12T18:50:08.503803Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.11232/citation-record","integrity":"/paper/2411.11232/integrity","json":"/paper/2411.11232/citation-record.json","paper":"/paper/2411.11232"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.902249Z","title":"Common ob- jective evaluation metrics, such as mel-cepstral distance (MCD)","venue":null,"work_id":"c982c004-fcc6-498d-98f2-508deee6ef91","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.253206Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:1d2e008f7def0a8ef2d397cf147017aef033fdee43e5aaec4906a627473f016e","observation_id":"a10d66e1-013c-4a02-a300-2a91e0e2666b","resolution":{"observed_at":"2026-08-12T18:50:08.906628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.888890Z","title":"As a result, some ob- jective measures or models related to human perception have been proposed [3, 4, 5, 6]","venue":null,"work_id":"7bedcb66-d445-4e8f-852a-27f0a5384a61","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.257531Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:061e8fd3f7d9215b5b22e9cc976ca976e22672998b357178aef7575ad5a97ec9","observation_id":"310222d8-cc72-43e2-8675-17cf92c5aab0","resolution":{"observed_at":"2026-08-12T18:50:08.893734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.853410Z","title":"Dataset In this paper, the experiments followed the same settings as the V oiceMOS Challenge 2022 [15]","venue":null,"work_id":"9224899e-e074-4127-a8d4-214976b97f5c","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.273361Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:60687b43b04e65ff99e41df85cf31d6ccd26a4e57fe0bc12f7b0988a34f9d57b","observation_id":"e87ed4fb-8977-4686-88e2-04c264b79dd2","resolution":{"observed_at":"2026-08-12T18:50:08.857226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.877118Z","title":"mean-listener","venue":null,"work_id":"6429d969-69dc-4bdd-b28d-71bd662c84ae","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.265344Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:25ac90a4056a10d7804d42f9c83e1dadf56d9b76c3346ae4cbf26b54ec4a3bc8","observation_id":"780f391c-0607-4a61-aa33-3478aefec9fb","resolution":{"observed_at":"2026-08-12T18:50:08.881238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.865387Z","title":"When the rater ID is not the mean-listener, the label representing the sample is the score given by the individual rater (an integer i from 1 to 5)","venue":null,"work_id":"57200f76-2173-4bed-b8ea-8e7c68383ab3","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.269386Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:4cd821e48383ec8aa994dda3839a0136064185cbefc0ca1bcc94a61a7a3e56ca","observation_id":"ee4f1d43-f25c-4638-bcb1-238c065bba6e","resolution":{"observed_at":"2026-08-12T18:50:08.869423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.758506Z","title":"NISQA: A deep CNN-self-attention model for multidimensional speech quality prediction with crowdsourced datasets,","venue":null,"work_id":"d2dc65cf-80b4-4adc-8444-668c21a0f9db","year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.308516Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:9055e7966ba82300b475b112a58eabefaa50c51f3400fcef1305be8d1b5339f8","observation_id":"1f5c6476-c622-4219-8e09-7e6423a05b1d","resolution":{"observed_at":"2026-08-12T18:50:08.762518Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.842105Z","title":"Comparision with baseline methods We first compare the proposed SAMOS with the baselines","venue":null,"work_id":"95a6bcbe-f631-49bd-88fd-3522583f1ff0","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.277306Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:8621773f8641cfc3bc35c45cb5769d269e0ee59f12ae0db9f43a2b74c58780b9","observation_id":"67bc5e51-1fd6-407d-bb72-5fb44d2d6e40","resolution":{"observed_at":"2026-08-12T18:50:08.846329Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.830352Z","title":"We can see that removing the semantic module resulted in the degradation of all the metrics on both datasets, indicating the importance of semantic repre- sentations from SSL model","venue":null,"work_id":"d0d41d79-7461-40ae-98b2-f7ec26a35fee","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.281046Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:28eafd28b6847a62b7e15d9d86a7ef487bfa2565ba3ccf27164a9a61f8ade9b2","observation_id":"41c32c20-0186-47f1-a6b0-8cb3bec56c93","resolution":{"observed_at":"2026-08-12T18:50:08.834338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.817291Z","title":"To improve prediction accuracy, SAMOS employs parallel regression and classification heads, and finally outputs the final MOS score through an aggregation layer","venue":null,"work_id":"df92a532-ab10-4e0a-8454-9ec2c7dba3ef","year":null},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.285155Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:b324e9d78652ee227fb7fa6a89a210b0307244e15f4e576df406dfd98dae802f","observation_id":"cdec2479-b405-4bf8-85b1-1d9413048786","resolution":{"observed_at":"2026-08-12T18:50:08.821712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.804285Z","title":"Mel-cepstral distance measure for objective speech quality assessment,","venue":null,"work_id":"fd22aadf-e09e-46df-924b-b7aeef0dbfcd","year":1993},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.288976Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:97be99fd05978067e77791b919073ae8364c5099c507ffe1cb6c75b15375eed1","observation_id":"a4a7314c-03f6-4e6b-9fbc-0c1eacc56cf3","resolution":{"observed_at":"2026-08-12T18:50:08.808890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"cited_work":{"arxiv_id":"2411.11232","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.11232","snapshot_observed_at":"2026-08-12T18:50:08.499324Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","venue":"cs.SD","work_id":"8e46c1f2-bd7d-4d79-bf58-d19ed1dcf6cc","year":2024},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.261278Z"},"links":{"cited_paper":"/paper/2411.11232","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:5681136c9e893c64e16ea39de0f2fb0fe04f40d1f8534ceae7a9eb5dc88d016c","observation_id":"9029bd3b-988a-4b5d-a2ce-96aa0b219744","resolution":{"observed_at":"2026-08-12T18:50:08.503803Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.292704Z","title":"SDR– half-baked or well done?","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.292704Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:c66c3508353cee8bc09d1b06e52669b937b8cb3cf3649b32619984f51732d578","observation_id":"30193a09-e986-462f-9cfc-2ca600386acb","resolution":{"observed_at":"2026-08-12T18:50:08.292704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.296480Z","title":"Per- ceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs,","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.296480Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:0e3d949743698ece6e32a8e429e5ea8681a23200d10994ca32276fa448d2fc62","observation_id":"e07f6f7b-3792-4d31-b07a-1e0954f8979d","resolution":{"observed_at":"2026-08-12T18:50:08.296480Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.300376Z","title":"A short- time objective intelligibility measure for time-frequency weighted noisy speech,","venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.300376Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:dcb1e13104322947b0dae058134c6df1f1ffc3ac62470c226f25e935c9804db2","observation_id":"e44b9181-7668-4052-8766-de5ac55a6734","resolution":{"observed_at":"2026-08-12T18:50:08.300376Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.770255Z","title":"ViSQOL: An objective speech quality model,","venue":null,"work_id":"7be79d1d-cfaa-4ba8-b87d-9ee747dc814b","year":2015},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.304873Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:32cbbc467fb08f6ffb3f6b98f49b40c4f38d56af0819e13b5d5a3cf19301cf7e","observation_id":"f594f578-5842-4e96-9dde-2ef897188c1c","resolution":{"observed_at":"2026-08-12T18:50:08.774239Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1611.09207","last_updated":"2016-11-28T15:51:25Z","snapshot_observed_at":"2026-08-14T21:28:18.830898Z","submitted_at":"2016-11-28T15:51:25Z","title":"AutoMOS: Learning a non-intrusive assessor of naturalness-of-speech","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1611.09207","snapshot_observed_at":"2026-08-12T18:50:08.312306Z","title":"AutoMOS: Learning a non- intrusive assessor of naturalness-of-speech,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.312306Z"},"links":{"cited_paper":"/paper/1611.09207","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:b215bdd7eb74c9623ec72494be3835f046704c53447712171fb0effe944ada10","observation_id":"e699b29c-92a3-4aad-8d43-3aa357453f6f","resolution":{"observed_at":"2026-08-12T18:50:08.312306Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.747475Z","title":"Quality-Net: An end-to-end non-intrusive speech quality assessment model based on blstm,","venue":null,"work_id":"a3f7a750-ff8d-4171-985d-c3288523c29e","year":2018},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.316397Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:09c1d41d760e0c277eb88705143d285a9055de276ce72b582a7ab03eda57a828","observation_id":"7b82dcd7-8716-4fbe-8279-c3b28f26b027","resolution":{"observed_at":"2026-08-12T18:50:08.751072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.320018Z","title":"MOSNet: Deep learning-based objec- tive assessment for voice conversion,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.320018Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:9da2e80ed7acf4e1dfbc6fd84f373fb01af268dba1c18a8fe90f8cc176d5f85d","observation_id":"3e78365f-4946-4647-898c-9c39a313e118","resolution":{"observed_at":"2026-08-12T18:50:08.320018Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.323526Z","title":"MBNet: MOS prediction for synthesized speech with mean-bias network,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.323526Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:524fe07738ef90e7da4574803728221f6c82b6aedfa6743640caac75d96c2fe1","observation_id":"3d77dae0-f23a-4757-844d-9fa4cf8e0cbe","resolution":{"observed_at":"2026-08-12T18:50:08.323526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.720787Z","title":"LDNet: Unified listener dependent modeling in mos prediction for syn- thetic speech,","venue":null,"work_id":"44e8ebae-0be2-4ebd-a58c-766f2f202c7c","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.326959Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:fb6c4b34d0611f71aa8f72dd22133b33566ee15df9715e4deafd865b3b5584d2","observation_id":"794cebee-6b36-4bbb-b50d-89a86f18a8e7","resolution":{"observed_at":"2026-08-12T18:50:08.724766Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.708305Z","title":"Generaliza- tion ability of mos prediction networks,","venue":null,"work_id":"c22b4a17-fe30-4c11-8f79-97ec27fbdbbf","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.330731Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:28e81181196d5a7653a9996a283172395f2caf1d5ea6f3f592fccaa8701d1780","observation_id":"52e2e90f-d821-48e3-befb-f3f91db3124b","resolution":{"observed_at":"2026-08-12T18:50:08.712368Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.695722Z","title":"Deep learning-based non-intrusive multi- objective speech assessment model with cross-domain features,","venue":null,"work_id":"a34e3bc5-2dbb-45e4-84cb-3b7edd27aa46","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.334617Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:19c36d459f09792a3f7c8afc8e69384549e917df9d9249a9714a96b8f1ead18d","observation_id":"fb553b37-1660-4ac6-83a8-6a2253627619","resolution":{"observed_at":"2026-08-12T18:50:08.700005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.337864Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.337864Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:2fd24fc64dceb070586dcc44e4380dd716aceea8b4c9809126bed47417d43bd3","observation_id":"60af0c23-252a-4d64-97a0-0496ea529157","resolution":{"observed_at":"2026-08-12T18:50:08.337864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.675662Z","title":"The V oiceMOS Challenge 2022,","venue":null,"work_id":"2a4d9ac0-5818-4c94-ab03-282a76865c44","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.341380Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:d15e8ec904c9d56a924d448898c93dba395e393069e6ea746b42c2460027e021","observation_id":"54fe6ef4-8be8-4958-8218-b5dc94710d9f","resolution":{"observed_at":"2026-08-12T18:50:08.679439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.664002Z","title":"A transfer and multi-task learning based approach for MOS predic- tion,","venue":null,"work_id":"14b26b01-8b32-4417-87ab-7d8a4eb3cde4","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.344950Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:735a71f78b7e7c3235b6705adea24ebb080e0d48d1b3526ad9bc5f5545195a7c","observation_id":"1c22bcfd-ac74-4282-8180-efc23733ff3e","resolution":{"observed_at":"2026-08-12T18:50:08.667934Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.652258Z","title":"DDOS: A MOS predic- tion framework utilizing domain adaptive pre-training and distri- bution of opinion scores,","venue":null,"work_id":"ec444b00-9197-4517-95f9-5c5bb2ee2004","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.348461Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:9cf39763417cf0204208e78858738dc2a1d65a4bdc34c48be2088ed742d5c03d","observation_id":"25448d39-b0e7-4b40-8abc-220ca5382bb6","resolution":{"observed_at":"2026-08-12T18:50:08.656349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.639687Z","title":"UTMOS: UTokyo-SaruLab system for V oice- MOS Challenge 2022,","venue":null,"work_id":"fdd48745-64f2-492f-9b55-4bf80565ed40","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.351975Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:7617211f7f2b869d293ff4186854d75e2ef2a68db5a9971453facc3d1c92c861","observation_id":"b4ec5239-740e-4ec7-912c-06563030db64","resolution":{"observed_at":"2026-08-12T18:50:08.643694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.627279Z","title":"Fusion of self-supervised learned models for MOS pre- diction,","venue":null,"work_id":"0dc7d1b9-351b-49dc-9fce-6ff1e61206e4","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.355250Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:c59290a223b561d9b8bf8524e18f551feef6b6ff9fbd46242f64d01f6f6c1ceb","observation_id":"c7776e9b-87ab-4c7b-9e29-10bdc810d981","resolution":{"observed_at":"2026-08-12T18:50:08.631520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.614236Z","title":"Ensem- ble of deep neural network models for MOS prediction,","venue":null,"work_id":"aaa9d78f-b3d0-4419-9909-ac8c58f3f217","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.359094Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:18d1ac53f3523287eebf8332db8fce5e2de5069f53376b58886a6d08827d9741","observation_id":"dded1169-beef-48c3-a257-46a1ba2d1be0","resolution":{"observed_at":"2026-08-12T18:50:08.618523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.601392Z","title":"RAMP: Retrieval- augmented MOS prediction via confidence-based dynamic weighting,","venue":null,"work_id":"d7809cef-e119-495f-9265-2f9b4da49f81","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.362859Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:5ec5a79442acf87e55d65100a1cf26cdae25bdbd530a4a7b4bce325fdefdce47","observation_id":"447a59fb-38f9-4672-a89c-98c72831c85d","resolution":{"observed_at":"2026-08-12T18:50:08.606030Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.588867Z","title":"Investigating content-aware neural text-to-speech MOS prediction using prosodic and linguistic fea- tures,","venue":null,"work_id":"2d053273-e285-4f57-9c35-4dc69c6998cf","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.366284Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:504334324e488b20182d87cc9f8a5a4a333e89a5ef62a72bcf4fcaf28de29a6f","observation_id":"91352e5e-3eba-47d5-ab95-306699f19a0c","resolution":{"observed_at":"2026-08-12T18:50:08.593262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.576237Z","title":"Wav2vec 2.0: A framework for self-supervised learning of speech repre- sentations,","venue":null,"work_id":"6a123a3b-2b7e-43a8-af7a-70c0a818f055","year":2020},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.369808Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:ab7fe735b59463f606787de9fdea8d4c879255184f042a88273d4fb57dd30d2d","observation_id":"1fee44e5-7b9f-45fb-8672-4163a1ba579f","resolution":{"observed_at":"2026-08-12T18:50:08.580675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02162","last_updated":"2024-06-04T09:51:02Z","snapshot_observed_at":"2026-08-19T21:30:55.567054Z","submitted_at":"2024-06-04T09:51:02Z","title":"BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation","version":1},"cited_work":{"arxiv_id":"2406.02162","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.02162","snapshot_observed_at":"2026-08-12T18:50:08.470377Z","title":"BiVocoder: A Bidirectional Neural Vocoder Integrating Feature Extraction and Waveform Generation","venue":"eess.AS","work_id":"43fc5d88-f339-4038-9ce1-49ba25d678d7","year":2024},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.373433Z"},"links":{"cited_paper":"/paper/2406.02162","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:39e47f92f540fff4b0abd92da08bfadf88486eff4d8478eb27ad4cebad3b4820","observation_id":"fffca9b7-87f0-4c05-9ff7-43ce8e77bfb5","resolution":{"observed_at":"2026-08-12T18:50:08.475207Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.565389Z","title":"SQAT-LD: Speech quality assessment transformer utilizing listener depen- dent modeling for zero-shot out-of-domain MOS prediction,","venue":null,"work_id":"e1e9ada1-e384-4688-a88d-3e7ff186d421","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.377396Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:a7d2bc14e96d8ec1e65f09e80f892801f1af3b7b63fbf6b1b9eea87cc29a6102","observation_id":"1027c81d-02da-479f-ac32-dcac8a1dbf17","resolution":{"observed_at":"2026-08-12T18:50:08.569287Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.02373","last_updated":"2021-06-30T05:45:26Z","snapshot_observed_at":"2026-08-19T19:12:45.921331Z","submitted_at":"2021-05-05T23:53:27Z","title":"How do Voices from Past Speech Synthesis Challenges Compare Today?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.02373","snapshot_observed_at":"2026-08-12T18:50:08.381069Z","title":"How do voices from past speech synthesis challenges compare today?","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.381069Z"},"links":{"cited_paper":"/paper/2105.02373","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:3f232141c7c34baaebddd9e710d5fd15f1e50f6bb432260cdfcd40781ade6de1","observation_id":"151e9e29-7922-43e7-982b-329479541b82","resolution":{"observed_at":"2026-08-12T18:50:08.381069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.385075Z","title":"The Blizzard Challenge 2019,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.385075Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:46205c6fc04ea265fac2dcde6f87eccfd1710148dd9767e1c23120be3ec410d2","observation_id":"02825d8f-0937-4a6c-8255-0eaafae92b5f","resolution":{"observed_at":"2026-08-12T18:50:08.385075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.389039Z","title":"Conformer: Convolution- augmented transformer for speech recognition,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.389039Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:74613b46d1edabe4087f824cd2b4b161b386d70e40a6af56d5a0cbc12ef0c963","observation_id":"3d4e407a-af34-485f-aa7f-04e59493193e","resolution":{"observed_at":"2026-08-12T18:50:08.389039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.539395Z","title":"ConvNeXt V2: Co-designing and scaling convnets with masked autoencoders,","venue":null,"work_id":"9af4c1be-61af-4e6e-86fd-a91700e11e6a","year":2023},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.393014Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:0204a4900b455346941e012c64cda3f0546fad82b85c59b65bdd1aa60b9f40ef","observation_id":"0fd6d85c-a83c-4c24-98d7-879b107e970d","resolution":{"observed_at":"2026-08-12T18:50:08.543400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.11030","last_updated":"2022-04-23T09:19:16Z","snapshot_observed_at":"2026-08-18T08:27:26.160969Z","submitted_at":"2022-04-23T09:19:16Z","title":"Improving Self-Supervised Learning-based MOS Prediction Networks","version":1},"cited_work":{"arxiv_id":"2204.11030","doi":null,"metadata_source":"pith","pith_arxiv_id":"2204.11030","snapshot_observed_at":"2026-08-12T18:50:08.439406Z","title":"Improving Self-Supervised Learning-based MOS Prediction Networks","venue":"eess.AS","work_id":"9682a3fd-5e48-4926-84ef-d28692b53d9d","year":2022},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.397560Z"},"links":{"cited_paper":"/paper/2204.11030","citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:094b7fe30d55355ed9ed5453926f7b93073ec134b2e85d8c7ffbab38b42689ef","observation_id":"632c41a8-ad18-440c-8953-02c144402482","resolution":{"observed_at":"2026-08-12T18:50:08.445952Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.526722Z","title":"ESP- Net: End-to-end speech processing toolkit,","venue":null,"work_id":"47de4001-bbf6-480d-8ee5-5090ddd83556","year":2018},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.402117Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:0ab62cc95beeae4daa30f2169fb902a76c340afddd6b0d78d1d9193700647071","observation_id":"567d4e74-2d84-44c6-a656-ca38ed976c65","resolution":{"observed_at":"2026-08-12T18:50:08.531013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T18:50:08.512667Z","title":"CSTR VCTK cor- pus: English multi-speaker corpus for CSTR voice cloning toolkit (version 0.92),","venue":null,"work_id":"ebf35385-736c-402a-86f0-3f5995b41afd","year":2019},"citing_paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T18:50:08.405900Z"},"links":{"citing_paper":"/paper/2411.11232"},"observation_digest":"sha256:15fd93c57b3569da8afb61509aa033a2670743f61c5a4e6d0cb317c4226b4dc7","observation_id":"b41c0c1c-fce8-4f73-bbbc-cc8666f5f234","resolution":{"observed_at":"2026-08-12T18:50:08.518311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.11232","last_updated":"2024-11-18T01:54:51Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-19T19:13:06.815342Z","submitted_at":"2024-11-18T01:54:51Z","title":"SAMOS: A Neural MOS Prediction Model Leveraging Semantic Representations and Acoustic Features"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":3,"verified_fuzzy":28},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 1 inbound Pith citation observation for arXiv:2411.11232."}