{"as_of":"2026-08-23T13:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:206c0c7ae29d7060fd7aafdf157426e3c57b40b9827f74ae75762d31a9566ebf","coverage":[{"denominator":38,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:50:02.022181Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:49:57.859145Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T14:50:02.310312Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"cited_work":{"arxiv_id":"2505.17446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17446","snapshot_observed_at":"2026-08-07T14:50:02.310312Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","venue":"cs.CL","work_id":"7adbccd3-3d74-4b74-bf5b-4e0a4a92d5a9","year":2025},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.859145Z"},"links":{"cited_paper":"/paper/2505.17446","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:b8a92b66363db5131b25f4938f3064236142e6a29cdb46e793a8718355307bfe","observation_id":"8cf812e2-ced0-49ae-874a-a2004ad39b8e","resolution":{"observed_at":"2026-08-07T14:50:02.409712Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.17446/citation-record","integrity":"/paper/2505.17446/integrity","json":"/paper/2505.17446/citation-record.json","paper":"/paper/2505.17446"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"cited_work":{"arxiv_id":"2505.17446","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.17446","snapshot_observed_at":"2026-08-07T14:50:02.310312Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","venue":"cs.CL","work_id":"7adbccd3-3d74-4b74-bf5b-4e0a4a92d5a9","year":2025},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.859145Z"},"links":{"cited_paper":"/paper/2505.17446","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:b8a92b66363db5131b25f4938f3064236142e6a29cdb46e793a8718355307bfe","observation_id":"8cf812e2-ced0-49ae-874a-a2004ad39b8e","resolution":{"observed_at":"2026-08-07T14:50:02.409712Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.415495Z","title":"Throughout this study, we used HuBERT [7] as an SSL model and extracted representations from the ninth layer","venue":null,"work_id":"7a3509e9-ef31-44fc-bb55-2b5cac99e785","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.887646Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:179b3f052f56a20f117cb6db4b7ae05cec1b57178789809f05fade96dfa746ff","observation_id":"0c1fb68c-fbc5-45de-b901-2f473c3febeb","resolution":{"observed_at":"2026-08-07T14:50:09.461214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.282625Z","title":null,"venue":null,"work_id":"08af9200-07e6-4497-b2bf-7f12dbf29b28","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:57.995598Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:d6e98ad88b9b8e4a9bade8d66baa14f552e1600f4f1fc829dd4da1805a30dae9","observation_id":"139ffd34-fb6c-4203-bc97-72dd762a371d","resolution":{"observed_at":"2026-08-07T14:50:09.330352Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.156762Z","title":null,"venue":null,"work_id":"26a31615-d59d-4669-ab64-51e3d7d17e8e","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.128986Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:5a614d4979bc6154a274b49c74d4806e97b1f0b9faabe7aac315678644286050","observation_id":"fe7a207b-24d1-47c7-aa0f-b692ededf434","resolution":{"observed_at":"2026-08-07T14:50:09.205464Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:09.032832Z","title":null,"venue":null,"work_id":"6dfafb07-f3c7-4068-aaae-3beb657972e5","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.238918Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:c34e5e355698cbb982104490881976d3cd0b873d41b3a5b4f7a55cc871729191","observation_id":"80c5be5a-d540-48db-a5fe-c97e88c5ee36","resolution":{"observed_at":"2026-08-07T14:50:09.076506Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.908184Z","title":"Dataset As a training set for SLM, we used LibriSpeech [17], a 960-hour English audiobook corpus","venue":null,"work_id":"d86554de-95bc-447e-9a24-dd31ce178874","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.300160Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:bad2744b76ebca93746cc224dd8fd5d58471a9090071d345642fcb8bd33e938b","observation_id":"6f133b16-38b6-4852-aea0-89f614884e69","resolution":{"observed_at":"2026-08-07T14:50:08.949260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.805177Z","title":"Figure 2 shows results on fixed boundary settings","venue":null,"work_id":"8068a7e2-544d-407b-ae3b-d1c8ce725be3","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.407640Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:f806f85ec2d94cb7d3c9f62043989c9d711dfc2da9fc7a0052fbbf3c289375e9","observation_id":"0845f3d5-355d-4e8a-974b-cb9c7422d12a","resolution":{"observed_at":"2026-08-07T14:50:08.846595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.667631Z","title":"yonder\"","venue":null,"work_id":"b7b7773d-e325-42c6-83b7-370a36a0328e","year":2016},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.481649Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:d235cd238985ee3e73be7cadc7acdb9c0369a9c2b694ba9a623f14b5acd1f14f","observation_id":"c225dd39-0b87-4151-a99c-c7a4a2ac90da","resolution":{"observed_at":"2026-08-07T14:50:08.730832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.534929Z","title":"We conducted mul- tiple speech tokenizations based on the combination of the fixed/variable segmentation and the cluster size","venue":null,"work_id":"7da09f53-d01b-4671-a821-1cc82fe0a3fa","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.570939Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:76af2fb569dcb584c527e06365077489129d6f485b367661c9dcf87ae52a17e7","observation_id":"564c9789-a8e0-4a74-943a-2d439fbab684","resolution":{"observed_at":"2026-08-07T14:50:08.591058Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.343235Z","title":null,"venue":null,"work_id":"0fd6642e-43b0-4d1b-9b45-8b032444c708","year":null},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.640221Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:7da25f8bab6eb904fb6ef8fe45c4f45c85652af7d1645e1133d2ea776c4c55e5","observation_id":"3fd089f2-4175-4795-b3fc-07373eed3187","resolution":{"observed_at":"2026-08-07T14:50:08.423077Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:08.107856Z","title":"On Generative Spoken Language Modeling from Raw Audio,","venue":null,"work_id":"bea2f7a1-3588-40c3-8fd6-a5f6fbdeb9c1","year":2021},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.753979Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:1eab99ae242977e5a0ad7ab7efc25a3e25ec415a38922d2e662d28c5afbcf003","observation_id":"87f1e936-e923-42bf-a50c-0b65c1b67fcc","resolution":{"observed_at":"2026-08-07T14:50:08.193845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.966936Z","title":"Textually Pretrained Speech Language Models,","venue":null,"work_id":"ca87f40a-afc0-4eb5-bb6f-bbfb4f503ef7","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.883691Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:1bc0ed13253a5c1bbce834c95031ed7797b589db705f98fa99591b88159ca454","observation_id":"70af3af0-d481-4544-949c-607c1f9a2449","resolution":{"observed_at":"2026-08-07T14:50:08.028899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.804572Z","title":"Audiolm: A language modeling approach to audio generation,","venue":null,"work_id":"50356337-eb6f-475b-8e82-d09ecb628bca","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:58.963630Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:4a312116da478eccaf48a2ac86a5f3993551e825e75b3754b36824919761c697","observation_id":"5a3393b4-1624-43bc-8954-f4f254b0e0ba","resolution":{"observed_at":"2026-08-07T14:50:07.884237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.660802Z","title":"WavLLM: Towards Robust and Adaptive Speech Large Language Model,","venue":null,"work_id":"a1dc5b7b-56f1-49fc-9c81-a98e352faeb3","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.048288Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:7243dc4a17a18a840b75c669946da02b792887a15d8ebfab5b108da30f5572d4","observation_id":"2a8b485e-ab94-43da-9934-9c0545480a8d","resolution":{"observed_at":"2026-08-07T14:50:07.713715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.484840Z","title":"Representation Learning with Contrastive Predictive Coding,","venue":null,"work_id":"5ce174db-a8af-447a-bfbc-63a5d2f45e89","year":2019},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.162114Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:8f42921c639b3ca71e656dc239b3bf6fd54f9a8bff18932fce27a2265c7b146c","observation_id":"3d90e8be-6ab2-4752-bf34-49cb88ce772e","resolution":{"observed_at":"2026-08-07T14:50:07.558127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.288707Z","title":"Wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Represen- tations,","venue":null,"work_id":"9c43d405-fa6d-4b6e-8d56-b47ba4962a83","year":2020},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.229667Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:2b6cc048625b6c0dcc7927c0c5e5b7097dbb03e0793cf0b96afa84440b41d305","observation_id":"399a0d2d-8807-4c88-86db-a7dbe180c12f","resolution":{"observed_at":"2026-08-07T14:50:07.379926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:07.047814Z","title":"HuBERT: Self-Supervised Speech Rep- resentation Learning by Masked Prediction of Hidden Units,","venue":null,"work_id":"b8f3ecca-e9be-4113-9910-35a5df53847d","year":2021},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.368364Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:fd874a3328cf688449a7c253751bff9b437e48c244cd1b3ec86ec5c1b41444ae","observation_id":"a2b91708-4c6d-450d-be67-836a8f2da1a2","resolution":{"observed_at":"2026-08-07T14:50:07.145029Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.820744Z","title":"The Zero Resource Speech Benchmark 2021: Metrics and baselines for unsupervised spoken language modeling,","venue":null,"work_id":"5ad3ec76-0e39-4a2d-a876-8337a4483bec","year":2021},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.502483Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:aa2f5624f68f05c6a623216a911d19d7e466e85712d67a102099469f4dfd6f98","observation_id":"5cb4db3e-8525-4a82-9f68-88280f53a06e","resolution":{"observed_at":"2026-08-07T14:50:06.928005Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.595210Z","title":"Generative Spoken Dialogue Language Model- ing,","venue":null,"work_id":"2fbf0db7-f464-4827-a08f-7070a0ac1a7b","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.637090Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:9a034f91dd15320a10e602289f31250ddfd8a9acb288c640d013522e75015a23","observation_id":"1bc6e752-38ab-46c8-b9ff-bbc5bdd885e5","resolution":{"observed_at":"2026-08-07T14:50:06.699670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.289082Z","title":"Direct Speech- to-Speech Translation With Discrete Units,","venue":null,"work_id":"07f189c4-47fb-4b0a-839e-c77fd8335c22","year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.792124Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:3c70753739a79a78656a509dd83edc947216646516800873b71550c2a9e4160f","observation_id":"0d1665fd-9f77-4ee3-8638-31813885e252","resolution":{"observed_at":"2026-08-07T14:50:06.423491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:06.036820Z","title":"Text-free prosody-aware generative spoken language modeling,","venue":null,"work_id":"96d9d5f1-ef14-4ce6-8e19-1af347896cb7","year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.930171Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:4be5291b0a237ced53b233276420d5559ed5cfafab476f894cba86f81e6e98c4","observation_id":"1c98c5ae-2af5-432e-a24b-6fe9e72d7ae4","resolution":{"observed_at":"2026-08-07T14:50:06.151214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.807236Z","title":"Attention is all you need,","venue":null,"work_id":"ae879658-0dd5-4230-bab2-8c8c1694a267","year":2017},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.108485Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:a02a6d15fa5f1ad6f7a0c4b2f83ee3bd2e061d2b08f48d84b8a39919006bf209","observation_id":"00f4e789-1c8c-404f-ace8-d8e74dbf0b7e","resolution":{"observed_at":"2026-08-07T14:50:05.922539Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.576279Z","title":"Self-Supervised Speech Representations are More Phonetic than Semantic,","venue":null,"work_id":"67c52bea-ad3f-4984-9de1-fa2ac894ec84","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.225350Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:647967b9da48dcd589e80dd3b415f1390f7a2259cff4d4e084134d9cfd5ecd61","observation_id":"2c6830ad-de9f-453a-9575-21f2fab6f352","resolution":{"observed_at":"2026-08-07T14:50:05.697064Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.296729Z","title":"Generative Spoken Language Model based on continuous word-sized audio tokens,","venue":null,"work_id":"53d2091e-b91d-47c7-bd88-84ef4383362f","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.329313Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:3bf1aeeb85103635111825de833057b65538d4cdd9b202791bb204430065676f","observation_id":"3639a5ac-d4ab-4078-8b27-57f965f6b5a0","resolution":{"observed_at":"2026-08-07T14:50:05.437143Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.04029","last_updated":"2024-10-05T04:29:55Z","snapshot_observed_at":"2026-08-16T13:12:23.643568Z","submitted_at":"2024-10-05T04:29:55Z","title":"SyllableLM: Learning Coarse Semantic Units for Speech Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04029","snapshot_observed_at":"2026-08-07T14:50:00.432842Z","title":"SyllableLM: Learn- ing Coarse Semantic Units for Speech Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.432842Z"},"links":{"cited_paper":"/paper/2410.04029","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:1aef101621647c60f1b55b9359212ee9a211445ce7103cc18e4bda59b41e97cb","observation_id":"45e2ab50-0b6a-4966-8c70-02191b2dff11","resolution":{"observed_at":"2026-08-07T14:50:00.432842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:05.030450Z","title":"Sylber: Syllabic Embedding Repre- sentation of Speech from Raw Audio,","venue":null,"work_id":"93ec52b2-da31-4fb9-a5af-04db88ad280a","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.556666Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:11a856911084de410f13cb1d5c429e722bad9874332f4533795e8b36b7a8db9e","observation_id":"4367e883-40a3-4b54-b078-4ae4544f7a68","resolution":{"observed_at":"2026-08-07T14:50:05.138108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:04.750264Z","title":"Lib- rispeech: An ASR corpus based on public domain audio books,","venue":null,"work_id":"16e04614-c90d-494b-b8ec-29ce3782eab1","year":2015},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.664857Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:1716ef7aeacde3febd6b3ff19893b7888222f94bc0fa6f68966cef3df500580e","observation_id":"4635ecbd-ac80-4160-9225-3d93d1ffc5be","resolution":{"observed_at":"2026-08-07T14:50:04.894664Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:04.476291Z","title":"Libri-light: A benchmark for ASR with limited or no super- vision,","venue":null,"work_id":"185abeff-2ec3-4725-8b60-aa73af36e0bb","year":2020},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.802807Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ec2d6c721d1f7ddd505354dc44bcc04107af7b1e2cc37f7947812caec730d63a","observation_id":"3efbaa60-489f-409b-80d8-cfd65dd5e073","resolution":{"observed_at":"2026-08-07T14:50:04.615415Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-07T14:50:00.957373Z","title":"OPT: open pre-trained transformer language models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:00.957373Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:efe604b1b3d975f56d31f90a914a64c083c5d6dc7edbe556a07578b9b4cab396","observation_id":"1e453aac-68e8-44d7-99a5-ce410e69dde7","resolution":{"observed_at":"2026-08-07T14:50:00.957373Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:04.231204Z","title":"ProsAudit, a prosodic benchmark for self-supervised speech models,","venue":null,"work_id":"e93a3e09-aa5f-4546-bb3b-cb264598cd2a","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.110052Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:a285c4ed077288b01bda1be8ad650c00c7207121c4cf20e3d5ef494594fe43aa","observation_id":"7877ab0c-d921-437a-85b6-25eff4f21cbc","resolution":{"observed_at":"2026-08-07T14:50:04.320994Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.993226Z","title":"A corpus and cloze evaluation for deeper understanding of commonsense stories,","venue":null,"work_id":"48528b60-61ee-440b-858e-8e57a7cb7850","year":2016},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.237648Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:c844924a57f891d9814e85aaeeb3afbb6d2d90ae7f96b38e89d15143cde804f8","observation_id":"688f3b9f-cfa1-49ac-9e5f-3dba2828f275","resolution":{"observed_at":"2026-08-07T14:50:04.110815Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.755553Z","title":"Praat: doing phonetics by com- puter [computer program]. version 6.4.27,","venue":null,"work_id":"49d8caaf-285c-47aa-8f43-13ed58f780c2","year":2025},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.363667Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:8fe92a6cec353ddbd14268786fb1a169ffb3145f21f0d699afebf03c4452eb05","observation_id":"5021f133-9058-4f65-ae6b-c35cee78d217","resolution":{"observed_at":"2026-08-07T14:50:03.857528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.482625Z","title":"Martinet, Elements of General Linguistics, ser","venue":null,"work_id":"c339d204-517d-4cc6-8d90-82ec1402f1c8","year":1966},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.513967Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:ddee085c6074870c7a179a91b53a2551c60774109b2038fdb1e2e2ca500101be","observation_id":"fc7b4e91-a9ec-48aa-a83e-67d0b3690784","resolution":{"observed_at":"2026-08-07T14:50:03.592334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.279265Z","title":"Are Discrete Units Nec- essary for Spoken Language Modeling?","venue":null,"work_id":"16bb16d1-3a0e-4dc5-a7f0-33fedd370949","year":2022},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.653076Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:9a14c6634471b1cb12334ee1f393774d5ea8e48fd6e5f114990b3d7747494fa9","observation_id":"3805a6d2-15f9-4b42-9db8-6599ed20e1da","resolution":{"observed_at":"2026-08-07T14:50:03.373953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.05755","last_updated":"2024-10-18T19:18:41Z","snapshot_observed_at":"2026-08-18T17:55:18.228697Z","submitted_at":"2024-02-08T15:39:32Z","title":"Spirit LM: Interleaved Spoken and Written Language Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.05755","snapshot_observed_at":"2026-08-07T14:50:01.779226Z","title":"Spirit- lm: Interleaved spoken and written language model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.779226Z"},"links":{"cited_paper":"/paper/2402.05755","citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:7af49dc1e8e5d57f81893846dad3eaf2d6860fc8a78fef9dfe6b886ce2781ea6","observation_id":"05e41d96-2af2-4730-a5b6-ac84e179efd6","resolution":{"observed_at":"2026-08-07T14:50:01.779226Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:03.020545Z","title":"Multi- resolution hubert: Multi-resolution speech self-supervised learn- ing with masked unit prediction,","venue":null,"work_id":"560ec706-bf39-406d-b687-d7ba4849ff6c","year":2024},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.870557Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:729550dce4941d77914d19780beebc528539fb8147c699daf75893a5ca4b6a09","observation_id":"47015db1-6fa3-4ab6-99af-e34b107b2ff2","resolution":{"observed_at":"2026-08-07T14:50:03.120389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:02.749152Z","title":"Self-supervised contrastive learning for unsupervised phoneme segmentation,","venue":null,"work_id":"ee3424f7-a887-4b69-88d2-cb117ed57e08","year":2020},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:01.956551Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:43a7f457669b2ffdfb0c369ff924987c96bb6e6361979c07d9d0584b2ee19690","observation_id":"a782d3a6-28f9-42e3-ba2e-e69c11c1d4c7","resolution":{"observed_at":"2026-08-07T14:50:02.865867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:50:02.541294Z","title":"Unsupervised word segmentation using temporal gradient pseudo-labels,","venue":null,"work_id":"9162935d-dad0-4aae-8ad0-731b455b77cc","year":2023},"citing_paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T14:50:02.022181Z"},"links":{"citing_paper":"/paper/2505.17446"},"observation_digest":"sha256:42811209c1d815113b1f92c22e8b932c3cf0835d99031dbc78b57a16505a4e75","observation_id":"abf54504-99ea-4786-be68-a93e630cec9a","resolution":{"observed_at":"2026-08-07T14:50:02.620676Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.17446","last_updated":"2025-05-31T13:32:13Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-18T17:55:56.319887Z","submitted_at":"2025-05-23T04:03:27Z","title":"Exploring the Effect of Segmentation and Vocabulary Size on Speech Tokenization for Speech Language Models"},"reference_resolution":{"displayed":38,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":7,"verified_exact":1,"verified_fuzzy":30},"total_outbound_references":38},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 38 of 38 outbound references and 1 inbound Pith citation observation for arXiv:2505.17446."}