{"as_of":"2026-08-10T06:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:aa5ac8b90ba827e621560443c2d59923554332b3d74107662da5913ac777282c","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-08T07:33:22.898615Z","state":"measured"},{"denominator":45,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":45,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-08T07:33:22.898615Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T07:34:43.130062Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"cited_work":{"arxiv_id":"2607.06392","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.06392","snapshot_observed_at":"2026-07-08T07:34:43.130062Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","venue":"cs.SD","work_id":"93fceddf-4119-4963-a7fa-016a6c5d3bd1","year":2026},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2607.06392","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9ab786e93fe92448709c14f04c9aa3f687cc81f66876a7815910616c983e4ecf","observation_id":"82848d1c-5f7d-4275-8eaa-c256a288cd70","resolution":{"observed_at":"2026-07-08T07:34:43.131579Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2607.06392/citation-record","integrity":"/paper/2607.06392/integrity","json":"/paper/2607.06392/citation-record.json","paper":"/paper/2607.06392"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.372864Z","title":null,"venue":null,"work_id":"9ec27f02-5dca-4216-b734-823b200b1b35","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:d90ab3aab33104032d1ade4b30884f68c2b3806221021cfe4fcd413bf38901c8","observation_id":"5e6941ad-25da-4c07-acd2-abeff45c5e65","resolution":{"observed_at":"2026-07-08T07:34:43.374441Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"cited_work":{"arxiv_id":"2607.06392","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.06392","snapshot_observed_at":"2026-07-08T07:34:43.130062Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","venue":"cs.SD","work_id":"93fceddf-4119-4963-a7fa-016a6c5d3bd1","year":2026},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2607.06392","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9ab786e93fe92448709c14f04c9aa3f687cc81f66876a7815910616c983e4ecf","observation_id":"82848d1c-5f7d-4275-8eaa-c256a288cd70","resolution":{"observed_at":"2026-07-08T07:34:43.131579Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.307113Z","title":"P” for predictive (masked prediction), “C","venue":null,"work_id":"257127dd-3bc6-448f-b278-d7f1ff4db3ca","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:21f9e1083547273f66df84756a87458f88ccc6ce5768e2f0966930b9f883d821","observation_id":"ed0d7673-27ed-4ba9-bc35-6a55de3b41b2","resolution":{"observed_at":"2026-07-08T07:34:43.309303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.387703Z","title":"flattens","venue":null,"work_id":"418e6a90-bf28-4583-98fb-7701a48de5bf","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:1c53c60d9b1b3f6d0967158c8b710eadf39127df6fef67737d3dcc5a5b6c2e3c","observation_id":"2de6934a-ee2a-4bbd-9f78-5582d959bb2e","resolution":{"observed_at":"2026-07-08T07:34:43.389772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.339658Z","title":"In terms of phonetic content (SpeechBERTScore), both SSL models display a broad region of high SpeechBERTScore values, indicating a stable phonetic encoding","venue":null,"work_id":"c0047eb9-c4a9-4a32-b1a7-b45b70ec0f58","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:5c21bf651eaa042c6daac922a11d3f2a6cd87ce40c4a8f25e3222ab8fe47fd78","observation_id":"77fa1bdf-2c72-49b1-b82b-e0aa07855210","resolution":{"observed_at":"2026-07-08T07:34:43.342170Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.343227Z","title":"Our analysis revealed dis- tinct optimization regimes, notably the late-stageentropy col- lapsein Wav2Vec2, contrasting with the geometric stability of WavLM","venue":null,"work_id":"19513fbc-0d34-4aef-8fcf-af913c61d088","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:ae147885455261ef4821a16d7684b932e4a4732741091e14c2f40c27da952fa9","observation_id":"9bec8b19-df44-4e57-96ff-b88bf026c7d0","resolution":{"observed_at":"2026-07-08T07:34:43.347145Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.319279Z","title":null,"venue":null,"work_id":"b2cc2478-1ca8-4bb5-833e-bf496dfec392","year":2025},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:36cfc898775bd80b0dcccd49d26c53995d73c40f261f73053e3bd441cc2afdba","observation_id":"27b71982-218e-42f6-a1a0-761dd69a6a5e","resolution":{"observed_at":"2026-07-08T07:34:43.322052Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.293105Z","title":"They did not contribute to the scientific content, analyses, or conclusions of this work","venue":null,"work_id":"b4f9750b-0861-438d-9c7c-1325dc7932d3","year":null},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:1b5b811ea8a319d6f8ee066fc5c3e54a477ae94b263bb20e124a0c801cd4fae2","observation_id":"35cb1050-b732-42c7-9696-4498b151b0d0","resolution":{"observed_at":"2026-07-08T07:34:43.295532Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.301897Z","title":"Audio self-supervised learning: A survey","venue":null,"work_id":"bdebf7c0-605c-40a4-8309-10c8cd5e56c9","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:2b78f6e81231d39fe07d64fbb936196b1bd28067fd999fa72e14f105e9d6537f","observation_id":"8e70f6de-0402-4dbd-8dd8-12f0d27e7eaf","resolution":{"observed_at":"2026-07-08T07:34:43.303685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.296267Z","title":"Self-supervised speech representation learning: A review","venue":null,"work_id":"2490870d-7d5c-4794-a031-fd48af7be202","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:76a638c05875814bf0ff50ed34edc8d14566e50fb45de58c6380f4c17ca775f3","observation_id":"dfbe4b0c-e9ee-4e7e-9929-612bc688dc09","resolution":{"observed_at":"2026-07-08T07:34:43.298028Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T12:16:14.563919Z","title":"Wavlm: Large-scale self- supervised pre-training for full stack speech processing","venue":null,"work_id":"1d47d864-93dd-41e6-9426-cfc627321f9f","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:3cbf3d77754ecc10e8af06c3de6198f514a10d1f1e708e37a9c008cc6b6a3aeb","observation_id":"bbe2d59d-b28a-451d-90d9-74bb0d110a6d","resolution":{"observed_at":"2026-07-08T07:34:43.335659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.393644Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech repre- sentations","venue":null,"work_id":"a3ebe5eb-bb98-4850-b76b-36c4083e687e","year":2020},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:b59886ae9cad49883598b5c5abd67a62d842b1706d4764e8f5c905deba047a56","observation_id":"b5720b33-bfa1-429f-8939-75636440af77","resolution":{"observed_at":"2026-07-08T07:34:43.396092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.404293Z","title":"Hubert: Self-supervised speech represen- tation learning by masked prediction of hidden units","venue":null,"work_id":"5e9a152d-1afa-42b6-b2a2-c022f7fce72a","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:8fa6d664a419980b23a8b9a231b2bd1cfe82aa28039f84a90ee3b4a6d9dd901f","observation_id":"b2204259-f1a5-4f4b-88be-e31f10c30167","resolution":{"observed_at":"2026-07-08T07:34:43.406451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2012.06185","last_updated":"2021-01-14T14:17:22Z","snapshot_observed_at":"2026-08-05T23:11:57.856234Z","submitted_at":"2020-12-11T08:22:23Z","title":"Exploring wav2vec 2.0 on speaker verification and language identification","version":2},"cited_work":{"arxiv_id":"2012.06185","doi":null,"metadata_source":"pith","pith_arxiv_id":"2012.06185","snapshot_observed_at":"2026-07-08T07:34:43.107921Z","title":"Exploring wav2vec 2.0 on speaker verification and language identification","venue":"cs.SD","work_id":"a5657a14-7f45-4606-9769-988459cbd727","year":2020},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2012.06185","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:2484662ee73d41c87c3434c9af804de8b9b1cb6aebd65042c6af40ae4264fbda","observation_id":"a363e606-192b-4193-9151-e4a170cc95c5","resolution":{"observed_at":"2026-07-08T07:34:43.109415Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.02735","last_updated":"2022-10-03T20:50:54Z","snapshot_observed_at":"2026-07-06T12:05:22.076764Z","submitted_at":"2021-11-04T10:39:06Z","title":"A Fine-tuned Wav2vec 2.0/HuBERT Benchmark For Speech Emotion Recognition, Speaker Verification and Spoken Language Understanding","version":3},"cited_work":{"arxiv_id":"2111.02735","doi":null,"metadata_source":"pith","pith_arxiv_id":"2111.02735","snapshot_observed_at":"2026-07-11T01:57:52.037436Z","title":"A fine-tuned wav2vec 2.0/hubert benchmark for speech emotion recognition, speaker verification and spoken language understanding","venue":"cs.CL","work_id":"6bb9f860-e45b-4ceb-8ca3-17d38124e800","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2111.02735","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:6aa37f2584442818ad1f756a5cbceea1fdb0c3b065a8a1530f3103f04ab59e5d","observation_id":"5890da5e-07d5-4cb0-a070-6e64d930fa15","resolution":{"observed_at":"2026-07-08T07:34:43.128546Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03502","last_updated":"2021-04-08T04:31:58Z","snapshot_observed_at":"2026-07-06T10:57:30.968546Z","submitted_at":"2021-04-08T04:31:58Z","title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","version":1},"cited_work":{"arxiv_id":"2104.03502","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.03502","snapshot_observed_at":"2026-07-08T07:34:43.114766Z","title":"Emo- tion recognition from speech using wav2vec 2.0 embeddings","venue":"cs.SD","work_id":"ebe96864-b6a5-48a9-9eee-f0502736a27f","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2104.03502","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:cd300405c505124149db3e2f4ce16814f7aaa380e1c5382e922c565c1b97782a","observation_id":"d94f2228-de5b-4331-8f5c-64da30fd43e3","resolution":{"observed_at":"2026-07-08T07:34:43.117158Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.377871Z","title":"Exploring wav2vec 2.0 fine tuning for improved speech emotion recognition,","venue":null,"work_id":"40a9f41a-ea94-407d-89a1-01806a5cbb37","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:b0651858fa5b88dff695082ab6fb7bad8119220bc8e337773a912a10776abb7b","observation_id":"f4d1bfe8-a277-4c42-ae3b-018a26f983d3","resolution":{"observed_at":"2026-07-08T07:34:43.379822Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.03339","last_updated":"2022-07-05T12:30:18Z","snapshot_observed_at":"2026-07-06T12:57:47.412026Z","submitted_at":"2022-04-07T10:22:26Z","title":"Boosting Self-Supervised Embeddings for Speech Enhancement","version":2},"cited_work":{"arxiv_id":"2204.03339","doi":null,"metadata_source":"pith","pith_arxiv_id":"2204.03339","snapshot_observed_at":"2026-07-08T07:34:43.133145Z","title":"Boosting Self-Supervised Embeddings for Speech Enhancement","venue":"eess.AS","work_id":"f1ce3a09-0b6c-460c-99a8-cb9087f20387","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2204.03339","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:22da0ba434b6ebae69a9b1b9aac65ca12a023973a32f6b51566fd7fd2436172d","observation_id":"942f4d2c-fab2-441b-b029-9e908a01ffef","resolution":{"observed_at":"2026-07-08T07:34:43.134667Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.326100Z","title":"Investigating self-supervised learning for speech enhancement and separation,","venue":null,"work_id":"a74e10e3-2660-4472-914c-973d9035dcac","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:34afb04f3a840cd22152f8668c7e93e5adbba2cd08931962e72990aed57ed8ab","observation_id":"dba035b9-43c3-4588-9e90-a248acc7674e","resolution":{"observed_at":"2026-07-08T07:34:43.328172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.310093Z","title":"Layer-wise analysis of a self-supervised speech representation model","venue":null,"work_id":"ffc62a6a-ca4d-4de1-a0b4-181aaffb684c","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:32e39a0fefea8618bae6baaf69854979a9aa38b47155a261266bd9f9a4295293","observation_id":"29d72771-7895-46ba-9604-4f6e0be24cc6","resolution":{"observed_at":"2026-07-08T07:34:43.311979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2101.00387","last_updated":"2021-07-12T22:46:37Z","snapshot_observed_at":"2026-08-07T20:40:16.363405Z","submitted_at":"2021-01-02T06:29:12Z","title":"What all do audio transformer models hear? Probing Acoustic Representations for Language Delivery and its Structure","version":2},"cited_work":{"arxiv_id":"2101.00387","doi":null,"metadata_source":"pith","pith_arxiv_id":"2101.00387","snapshot_observed_at":"2026-07-08T07:34:43.100777Z","title":"What all do audio transformer models hear? probing acoustic rep- resentations for language delivery and its structure","venue":"cs.CL","work_id":"2dd54a11-e44d-4b89-913f-ca5c09c31009","year":2021},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2101.00387","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:a8bc08db9cd2d7dcf518c68f81f50cb2e4baa7745f67852ebfb2d6c1f1835fa9","observation_id":"36d8a162-c6e8-4401-98cf-c998e26c91d3","resolution":{"observed_at":"2026-07-08T07:34:43.103019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.299014Z","title":"Phone and speaker spatial organization in self-supervised speech represen- tations,","venue":null,"work_id":"53bbe33f-b213-4d47-a8f4-61d3f51034a6","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:a8dd836615d9a467cbab23260c2693dd296baa38f1fc484a659d814296c2283c","observation_id":"7c469199-a373-4f13-8db3-5d484cd9d45f","resolution":{"observed_at":"2026-07-08T07:34:43.301070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.383869Z","title":"Exploration of a self- supervised speech model: A study on emotional corpora,","venue":null,"work_id":"2c26a125-14de-427c-8e23-f78d1b9a65e5","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:1283d4371a47a91df32cf8a1ebc4883be7a3e1d32df4739bc10ac1d7091e3102","observation_id":"813eae8a-21ba-4a2f-b6f2-aa9d57a6c14d","resolution":{"observed_at":"2026-07-08T07:34:43.386830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.401883Z","title":"What do self-supervised speech and speaker mod- els learn? new findings from a cross model layer-wise analysis,","venue":null,"work_id":"1d09a53c-348b-4b4a-b2bc-a9d0e2829407","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9962ae9be441242f4836a890500d0ab24af62c3df506602ffcb2eb99f0076b6d","observation_id":"8fbb7544-e92a-44e3-8ddf-d03abc8b4269","resolution":{"observed_at":"2026-07-08T07:34:43.403501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.397202Z","title":"Comparative layer-wise anal- ysis of self-supervised speech models,","venue":null,"work_id":"d3e629f2-aa10-4fb8-a634-93d3ae24ffe8","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:b8087a1fa967c1edec1f7ab73884c5c59f147a51effab2b842530fe1b988267d","observation_id":"795b1293-8c35-481f-ae1f-491ace693320","resolution":{"observed_at":"2026-07-08T07:34:43.399306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2501.05310","doi":"10.48550/arxiv.2501.05310","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A large-scale probing analysis of speaker-specific at- tributes in self-supervised speech representations,","venue":"ArXiv.org","work_id":"d3f2fac3-7f36-4861-b473-562f66ebf28c","year":2025},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:b017673bc64ec38217091272602b7c619ca72e0f62195ebea0e7221215401a6e","observation_id":"06e38de1-db96-468e-a588-fec3d33cf9dd","resolution":{"observed_at":"2026-07-08T07:34:43.099226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02013","last_updated":"2025-06-15T18:27:17Z","snapshot_observed_at":"2026-08-01T16:50:50.645857Z","submitted_at":"2025-02-04T05:03:42Z","title":"Layer by Layer: Uncovering Hidden Representations in Language Models","version":2},"cited_work":{"arxiv_id":"2502.02013","doi":"10.48550/arxiv.2502.02013","metadata_source":"pith","pith_arxiv_id":"2502.02013","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Layer by Layer: Uncovering Hidden Representations in Language Models","venue":"cs.LG","work_id":"7b4ac06a-e804-4f0a-8305-c45f2735afb5","year":2025},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2502.02013","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:ffa638f5b1f8b348a740f9728b1281c31f9114a8a16ce55b0352c416e41e7b2d","observation_id":"75a8d211-9522-437a-96ad-26166bbd215e","resolution":{"observed_at":"2026-07-08T07:34:43.095089Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"physics/0004057","last_updated":"2000-04-24T15:22:30Z","snapshot_observed_at":"2026-08-07T10:46:08.274157Z","submitted_at":"2000-04-24T15:22:30Z","title":"The information bottleneck method","version":1},"cited_work":{"arxiv_id":"physics/0004057","doi":"10.48550/arxiv.physics/0004057","metadata_source":"pith","pith_arxiv_id":"physics/0004057","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The information bottleneck method","venue":"physics.data-an","work_id":"72655a80-0724-45ad-a330-1f4ed7aa613b","year":2000},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/physics/0004057","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:fb9f61f2d332860b4647e1824eec45aecc5b77245c9487dbdcca3679da5f2890","observation_id":"6e844674-e6a0-4843-9c64-ee80b6a53b5c","resolution":{"observed_at":"2026-07-08T07:34:43.091656Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-07-10T23:49:06.793856+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T23:49:06.793856+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1703.00810","last_updated":"2017-04-29T17:32:47Z","snapshot_observed_at":"2026-08-08T16:36:13.828301Z","submitted_at":"2017-03-02T14:53:14Z","title":"Opening the Black Box of Deep Neural Networks via Information","version":3},"cited_work":{"arxiv_id":"1703.00810","doi":"10.48550/arxiv.1703.00810","metadata_source":"pith","pith_arxiv_id":"1703.00810","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Opening the Black Box of Deep Neural Networks via Information","venue":"cs.LG","work_id":"3b14f412-2206-469d-bfd3-c387c75ea711","year":2017},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/1703.00810","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:563a0d4298bc0b307c316448127897d52734d77dd01307f5cac6a2db516cd62c","observation_id":"b23f0263-c011-4454-b4c2-d61de8dd7a7b","resolution":{"observed_at":"2026-07-08T07:34:43.106484Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-07-11T19:50:18.979424+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T19:50:18.979424+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.407415Z","title":"To compress or not to compress—self-supervised learning and information theory: A review,","venue":null,"work_id":"fe601810-66ee-493b-9e1f-9d606af72802","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9b171e43f3b2692d0fe81329f613e78b2de0074a49d8ee53c9a35ed603f20256","observation_id":"a152c60a-aefe-49da-a0c1-d8296a74b34d","resolution":{"observed_at":"2026-07-08T07:34:43.409306Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.348159Z","title":"Large language models implic- itly learn to straighten neural sentence trajectories to construct a predictive representation of natural language","venue":null,"work_id":"0b7fd58b-0346-4ccd-bec4-ee1f9f6524ce","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:4852af6a13caad316d019961c6823e5b89febea87c70653847718ca9e70817b8","observation_id":"cdc9bef5-c658-48ea-86a5-33b26788c949","resolution":{"observed_at":"2026-07-08T07:34:43.350133Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1807.03748","last_updated":"2019-01-22T18:47:12Z","snapshot_observed_at":"2026-07-06T06:49:24.960992Z","submitted_at":"2018-07-10T16:52:11Z","title":"Representation Learning with Contrastive Predictive Coding","version":2},"cited_work":{"arxiv_id":"1807.03748","doi":"10.1609/aaai.v36i10.21390","metadata_source":"pith","pith_arxiv_id":"1807.03748","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Representation Learning with Contrastive Predictive Coding","venue":"cs.LG","work_id":"7b08a1d4-d565-424e-9c86-6ef244b7b90a","year":2018},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/1807.03748","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:61c8a5015c1f9250b0c5f681aa36780e413cd6e647b0bc70e81d16a4878afa56","observation_id":"48ddab6f-1fa6-47d6-80ea-9b1d26417afc","resolution":{"observed_at":"2026-07-08T07:34:43.085386Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04000","last_updated":"2023-12-07T02:31:28Z","snapshot_observed_at":"2026-08-01T16:37:51.615090Z","submitted_at":"2023-12-07T02:31:28Z","title":"LiDAR: Sensing Linear Probing Performance in Joint Embedding SSL Architectures","version":1},"cited_work":{"arxiv_id":"2312.04000","doi":"10.48550/arxiv.2312.04000","metadata_source":"pith","pith_arxiv_id":"2312.04000","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2312.04000 , year=","venue":"cs.LG","work_id":"3f4999db-6f36-4b5d-a91c-d07f2c21a2ee","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2312.04000","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:dcbdb2bac863f772bbbd28ed1252f5ba802305741af61ee61b4d3815415f80a2","observation_id":"38a4f022-4b99-4fb5-9bf3-890ef975b247","resolution":{"observed_at":"2026-07-08T07:34:43.125210Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.08164","last_updated":"2023-07-27T18:26:45Z","snapshot_observed_at":"2026-08-06T18:44:25.708968Z","submitted_at":"2023-01-19T16:56:21Z","title":"DiME: Maximizing Mutual Information by a Difference of Matrix-Based Entropies","version":3},"cited_work":{"arxiv_id":"2301.08164","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.08164","snapshot_observed_at":"2026-07-08T07:34:43.110807Z","title":"arXiv preprint arXiv:2301.08164 , year=","venue":"cs.LG","work_id":"2645d62b-e608-4019-b93e-c8fe50ba552d","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2301.08164","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:a215c162714ae8f53d0a2733dec203a8ca93b338aab8c340ac2096fcc417c789","observation_id":"2c331c9a-a757-4979-8678-2bdeb9a338a5","resolution":{"observed_at":"2026-07-08T07:34:43.112949Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.369308Z","title":"Multivari- ate extension of matrix-based r ´enyi’s alpha-order entropy func- tional,","venue":null,"work_id":"6f08b0e1-7005-4e8a-9809-152de99d8ea0","year":2019},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:66e8aac8999ab94043be5b5bbe9020e3cdd7079727ecfba0a8fc82b2376ba027","observation_id":"120c595d-e87d-4669-b885-91e85b5d808e","resolution":{"observed_at":"2026-07-08T07:34:43.372137Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.356875Z","title":"Data2vec: A general framework for self-supervised learning in speech, vision and language,","venue":null,"work_id":"0cc494b7-0332-4cf3-91e3-354527645f16","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:ad1e82e2f8b46b09f38feaa4933d67f78ef0ece5246d8f07fa65e507a51484a0","observation_id":"e9e23925-46c6-4ea7-9ffe-5215a8b695ff","resolution":{"observed_at":"2026-07-08T07:34:43.362731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.304509Z","title":"Lib- rispeech: an asr corpus based on public domain audio books","venue":null,"work_id":"56dd2083-75da-478e-a91d-6aef8a2937a4","year":2015},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:9eb3c72711983765fa9cf0e7a4661757ed28405fd8e63de5290019bdb9794963","observation_id":"3f48d494-06d2-40d1-b856-621afba09e87","resolution":{"observed_at":"2026-07-08T07:34:43.306218Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.02747","last_updated":"2023-02-08T15:46:05Z","snapshot_observed_at":"2026-08-02T18:24:58.914589Z","submitted_at":"2022-10-06T08:32:20Z","title":"Flow Matching for Generative Modeling","version":2},"cited_work":{"arxiv_id":"2210.02747","doi":"10.1038/s41467-024-47656-z","metadata_source":"pith","pith_arxiv_id":"2210.02747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Flow Matching for Generative Modeling","venue":"cs.LG","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","year":2022},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2210.02747","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:8559c530eb820190fd97c8b496785972287550e404d1f6ec6a18f368025242ec","observation_id":"62b502cf-d6d2-4084-80c8-278417492d17","resolution":{"observed_at":"2026-07-08T07:34:43.121432Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"crossref_status_cache"},{"observed_at":"2026-06-01T22:57:59.860918+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.363806Z","title":"Scalable diffusion models with transform- ers","venue":null,"work_id":"30e140cc-f93d-4815-9180-92379f790636","year":2023},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:803ba3075188940a154fdb05e7190eda795be9faab1b2089061bfd2fbe2be5fd","observation_id":"89897661-0cd8-492d-933a-242e58106ef7","resolution":{"observed_at":"2026-07-08T07:34:43.365662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.336708Z","title":"Hifi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis,","venue":null,"work_id":"a0b0841b-3ae3-4344-8bb9-c83e91447e68","year":2020},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:22d2ea5e11b7224b66cdd208106a115de84dc908e4c334121fb03d452441cca6","observation_id":"19569968-b82b-4ed5-83a7-388dc37522a7","resolution":{"observed_at":"2026-07-08T07:34:43.338700Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16812","last_updated":"2024-09-01T14:34:27Z","snapshot_observed_at":"2026-08-06T17:12:30.832433Z","submitted_at":"2024-01-30T08:26:28Z","title":"SpeechBERTScore: Reference-Aware Automatic Evaluation of Speech Generation Leveraging NLP Evaluation Metrics","version":3},"cited_work":{"arxiv_id":"2401.16812","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.16812","snapshot_observed_at":"2026-07-08T07:34:43.086784Z","title":"SpeechBERTScore: Reference-aware automatic evaluation of speech generation leveraging nlp evaluation metrics","venue":"cs.SD","work_id":"8d315452-24ee-4b4a-befa-d194323ff285","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"cited_paper":"/paper/2401.16812","citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:5d8ab7d4dc7dade281c4fddf0f0ace1be07906a242aeae18008bcb1f43defaaa","observation_id":"69b0a9c6-e15d-42b4-901c-aa49d03539e9","resolution":{"observed_at":"2026-07-08T07:34:43.088448Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.312848Z","title":"Generalized end- to-end loss for speaker verification,","venue":null,"work_id":"09d3ea9b-8847-4901-8d91-ab2564c8e681","year":2018},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:14feb3ca155f67e24122eef175357b8e3acaf869879a29bd1df69af6f7c23ee2","observation_id":"01c1c60e-7173-4b4b-9790-0160e0efcb2e","resolution":{"observed_at":"2026-07-08T07:34:43.317506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.375243Z","title":"High-fidelity neural phonetic posteriorgrams,","venue":null,"work_id":"1379fb12-f3e5-40c4-b95c-5c360375d986","year":2024},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:5ad028b8acd299954d98bb1d6344d4b135db8de7cdc8d6e3914c1cb9089f8f79","observation_id":"b5c06f6a-c24f-4f81-8d89-b89ba7dfbb1b","resolution":{"observed_at":"2026-07-08T07:34:43.376922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T07:34:43.366713Z","title":"Crepe: A convo- lutional representation for pitch estimation,","venue":null,"work_id":"f2c16c71-198a-4501-bb40-e6c5764a7d15","year":2018},"citing_paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-08T07:33:22.898615Z"},"links":{"citing_paper":"/paper/2607.06392"},"observation_digest":"sha256:f6d4fba020dafa8c114d87b6c3053dd5b3ce077a28ec3b94704f88930c308b16","observation_id":"e1ba50e0-15ff-4b64-ba78-c9419ee83a97","resolution":{"observed_at":"2026-07-08T07:34:43.368407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.06392","last_updated":"2026-07-07T15:25:47Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-05T23:10:43.933121Z","submitted_at":"2026-07-07T15:25:47Z","title":"InsideSSL: Understanding Self-Supervised Speech Representations using a Model-Centric Perspective"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":0,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":2,"verified_exact":13,"verified_fuzzy":27},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 1 inbound Pith citation observation for arXiv:2607.06392."}