{"as_of":"2026-08-13T10:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:500f094a832713d38051afe946ecd24981cda6f645e5c0aee78ce6034999700c","coverage":[{"denominator":35,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:17:18.443046Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T16:17:14.961641Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-01T21:36:15.429731Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-08-04T16:17:14.961641Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:14.961641Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:438e2f442788eb267137fc98eb59fe2b50b3a5220479b72ef25c94c0b3dc494a","observation_id":"197a8495-d4f5-40e1-b4bf-2d54ab5e1cd0","resolution":{"observed_at":"2026-08-04T16:17:14.961641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":"2509.15001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-07-01T21:36:15.429731Z","title":"Babyhubert: Multilingual self-supervised learning for segmenting speakers in child-centered long-form recordings","venue":"eess.AS","work_id":"4b68d120-3d68-453c-9053-68524cb61b19","year":2025},"citing_paper":{"arxiv_id":"2605.19130","last_updated":"2026-05-18T21:30:54Z","snapshot_observed_at":"2026-08-01T19:57:38.697860Z","submitted_at":"2026-05-18T21:30:54Z","title":"EgoBabyVLM: Benchmarking Cross-Modal Learning from Naturalistic Egocentric Video Data","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T12:12:45.251924Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2605.19130"},"observation_digest":"sha256:d901cc697a88748ce1ab03243cf70cc40819071511b1e359c88cb356c591a0d7","observation_id":"93e9ca60-a992-4c37-9e85-87fe1977cb74","resolution":{"observed_at":"2026-06-30T03:17:23.577707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":"2509.15001","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-07-01T21:36:15.429731Z","title":"Babyhubert: Multilingual self-supervised learning for segmenting speakers in child-centered long-form recordings","venue":"eess.AS","work_id":"4b68d120-3d68-453c-9053-68524cb61b19","year":2025},"citing_paper":{"arxiv_id":"2606.01134","last_updated":"2026-05-31T10:12:47Z","snapshot_observed_at":"2026-08-06T21:55:11.856032Z","submitted_at":"2026-05-31T10:12:47Z","title":"Context-aware child-directed speech detection from long-form recordings","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T16:41:39.242748Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2606.01134"},"observation_digest":"sha256:bd415f6c6f677d44e75fc3082d51f6257fa04a30d388c205bbbe64d08ac4ce03","observation_id":"32db9e29-730a-4637-ade3-0a6bc1b8ff2a","resolution":{"observed_at":"2026-07-01T21:36:15.431349Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-07-12T04:11:03.723969Z","title":"Available: https://arxiv.org/abs/2509.15001","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03201","last_updated":"2026-07-03T11:14:54Z","snapshot_observed_at":"2026-08-10T03:58:59.586200Z","submitted_at":"2026-07-03T11:14:54Z","title":"Deriving Benchmarking Datasets from Long-Form Recordings: Challenges and Opportunities","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-12T04:11:03.723969Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2607.03201"},"observation_digest":"sha256:2602e4c4307bdd12aca6048f30d2b2cc7b0fe2ce0905e57536b2210c94015a46","observation_id":"888dcef7-0ea1-4f2f-afb2-793b987d7be0","resolution":{"observed_at":"2026-07-12T04:11:03.723969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.15001/citation-record","integrity":"/paper/2509.15001/integrity","json":"/paper/2509.15001/citation-record.json","paper":"/paper/2509.15001"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:14.871859Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:14.871859Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:4c4987a71f7c4ae604cb314d69c251d0611475d640411571894b1e4a589ee1ac","observation_id":"7de6dd56-37e1-4350-8836-46d899941921","resolution":{"observed_at":"2026-08-04T16:17:14.871859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.15001","snapshot_observed_at":"2026-08-04T16:17:14.961641Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:14.961641Z"},"links":{"cited_paper":"/paper/2509.15001","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:438e2f442788eb267137fc98eb59fe2b50b3a5220479b72ef25c94c0b3dc494a","observation_id":"197a8495-d4f5-40e1-b4bf-2d54ab5e1cd0","resolution":{"observed_at":"2026-08-04T16:17:14.961641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.025234Z","title":"Datasets Our pre-training dataset comprises 19 diverse datasets spanning mul- tiple continents over 40 languages (see Table 1)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.025234Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:be2bd867bfb20ee83ee7bfb71a678f04367ceaa066f5fcd8b88f06273ba08429","observation_id":"face384d-4aea-4164-912d-416f698c3169","resolution":{"observed_at":"2026-08-04T16:17:15.025234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.173475Z","title":"BabyHuBERT-2 achieves 64.0% average F-score, substan- tially outperforming both W2V2-LL4300 (58.7%) and HuBERT base (51.4%)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.173475Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:ef7a5d9f8f9cbe80efa54e91dfaefe444d4f7b4ea2ec248afc40a1a2fda74907","observation_id":"510ceb85-f257-4c63-b636-4d5d40e6f4d3","resolution":{"observed_at":"2026-08-04T16:17:15.173475Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.251303Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.251303Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:ac5d7f1683b8d4a71db152859a880a9b143c20571b7da31ae00870b4fbcb3e23","observation_id":"5b1683b2-2c2b-424a-85c5-e4e61d791237","resolution":{"observed_at":"2026-08-04T16:17:15.251303Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.319899Z","title":"ED: ERC (InfantSimulator); AC and TK: ERC (ExELang, 101001095)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.319899Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:bfb54fc58fadbf8f6a32067592c6df39153d01981b8fc2217480d2258c901ded","observation_id":"f79fc0c0-83aa-40a2-8a68-5c153a43870c","resolution":{"observed_at":"2026-08-04T16:17:15.319899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.04710","last_updated":"2025-03-06T18:57:16Z","snapshot_observed_at":"2026-08-10T05:06:53.825761Z","submitted_at":"2025-03-06T18:57:16Z","title":"Self-Supervised Models for Phoneme Recognition: Applications in Children's Speech for Reading Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.04710","snapshot_observed_at":"2026-08-04T16:17:15.909983Z","title":"Self-supervised models for phoneme recognition: Applica- tions in children’s speech for reading learning,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.909983Z"},"links":{"cited_paper":"/paper/2503.04710","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:5dffe3465a88bbdd8cc3f70e3128ce92a621c1acf542060d3bce72c3d3a66707","observation_id":"7c306f31-f294-46c4-b7e2-7497abf74319","resolution":{"observed_at":"2026-08-04T16:17:15.909983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.375905Z","title":"Long-form recordings to study children’s language input and output in under-resourced contexts,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.375905Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:1585109fb442d05be6b203690f36b000246b7a2c0e925f4ae2be3e1370b39c58","observation_id":"5ebe4f6e-9d68-4e01-9ef2-c39762f82b1d","resolution":{"observed_at":"2026-08-04T16:17:15.375905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.11075","last_updated":"2025-06-04T01:45:42Z","snapshot_observed_at":"2026-08-10T20:21:44.786225Z","submitted_at":"2025-06-04T01:45:42Z","title":"Fifteen Years of Child-Centered Long-Form Recordings: Promises, Resources, and Remaining Challenges to Validity","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.11075","snapshot_observed_at":"2026-08-04T16:17:15.475230Z","title":"Fifteen years of child-centered long-form recordings: Promises, resources, and remaining challenges to validity,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.475230Z"},"links":{"cited_paper":"/paper/2506.11075","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:6202154db0e89bcb9e6b01d1a75b7e9d554e3c33defa64257dda7831bcb21167","observation_id":"202ab127-9cbc-4509-b055-e5bc040dff58","resolution":{"observed_at":"2026-08-04T16:17:15.475230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.594105Z","title":"Acoustics of children’s speech: Developmental changes of temporal and spectral parameters,","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.594105Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:9e85568375f09e1f043af8517b03e3541301e614a8c0ce9d6033101d5b96f8d0","observation_id":"b8c07514-1220-41b0-a806-cedd3c2526c1","resolution":{"observed_at":"2026-08-04T16:17:15.594105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.686273Z","title":"Acous- tic variability and automatic recognition of children’s speech,","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.686273Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:152e5e2df73d17e528102794e1584c4c016066dbb9c13f901836aa47cede2afc","observation_id":"4fce2812-16b9-47ff-a12f-dd9e1ec3d8cb","resolution":{"observed_at":"2026-08-04T16:17:15.686273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.06733","last_updated":"2021-10-13T14:03:07Z","snapshot_observed_at":"2026-08-02T04:37:32.946853Z","submitted_at":"2021-10-13T14:03:07Z","title":"Systematic Inequalities in Language Technology Performance across the World's Languages","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.06733","snapshot_observed_at":"2026-08-04T16:17:15.777749Z","title":"Systematic inequalities in language technology performance across the world’s languages,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.777749Z"},"links":{"cited_paper":"/paper/2110.06733","citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:52e8dc5aabc7564e886006db5e92e87c0b4bdc6b4a489aff6a7faa3fc170a138","observation_id":"0300132c-f066-434f-b5b7-a68eba132b49","resolution":{"observed_at":"2026-08-04T16:17:15.777749Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.846220Z","title":"Introduction To Partial Fine-tuning: A Comprehensive Evaluation Of End-to-end Chil- dren’s Automatic Speech Recognition Adaptation,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.846220Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:772f95f6dd18e7c0f1a6d47eab64c5484c93cff618c201bc238160ed802e5abe","observation_id":"988e1431-a2c6-4edc-8a5e-6a8d7b2284f9","resolution":{"observed_at":"2026-08-04T16:17:15.846220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.645092Z","title":"A thorough eval- uation of the language environment analysis (lena) system,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.645092Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:1190bad34339136ec235d7bc8900316c0822605b668ecd4708585317c1d2576f","observation_id":"08868511-8a62-416a-969a-e319ae95028c","resolution":{"observed_at":"2026-08-04T16:17:16.645092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.065134Z","title":"Towards robust family-infant audio analysis based on unsu- pervised pretraining of wav2vec 2.0 on large-scale unlabeled family audio,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.065134Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:e92d3c08a45c93b7cb6a9b9f6244209118d4d766a8997a4e768c76e145726781","observation_id":"dfa272ec-0ca2-46d5-aed5-fba243dff1c1","resolution":{"observed_at":"2026-08-04T16:17:16.065134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.163714Z","title":"wav2vec 2.0: A framework for self-supervised learning of speech representa- tions,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.163714Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:db0f7dad5dc04857fc5d02e7bacb03a6916f55e90526830a6b328318b030dbdd","observation_id":"feb48427-a221-43cd-8f5f-8576c170fbf7","resolution":{"observed_at":"2026-08-04T16:17:16.163714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.231220Z","title":"Employing self- supervised learning models for cross-linguistic child speech maturity classification,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.231220Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:c7f785ca59223dda4be03561fef9d422668f05761dd59cc2c7d44892afd7b8b9","observation_id":"029aa4a9-e35c-4fa9-8e01-6fa19d4b96bb","resolution":{"observed_at":"2026-08-04T16:17:16.231220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.293783Z","title":"Reverse en- gineering language acquisition with child-centered long-form recordings,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.293783Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:0d247b9c5a5a8ec2224021e8fa7e32b14634251e51e02d6ab8119ef41cb588e5","observation_id":"f172a12f-aa0d-43f6-9edf-0b122c1bb1eb","resolution":{"observed_at":"2026-08-04T16:17:16.293783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.386578Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.386578Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:f044cb872c3070399b1d4dedd7c88a69c1cbadbba60d6ff772b812527985ebbf","observation_id":"a1b580f4-1acc-4e78-97cf-280634a1b5f1","resolution":{"observed_at":"2026-08-04T16:17:16.386578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.495794Z","title":"Homebank: An online repository of daylong child-centered audio recordings,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.495794Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:e8554c2fc762293379c7c566b1b1e4ac56ca1f18e990ccf45fe7f6abd8e53c73","observation_id":"55a9fde8-e000-4f5a-afb9-a854368c5971","resolution":{"observed_at":"2026-08-04T16:17:16.495794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.549676Z","title":"mhubert-147: A compact multilingual hubert model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.549676Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:54b49f98c7b5a1e7d03cea67825ac7db53fc2094be0b50b0da42d9242dbbc3e6","observation_id":"8107da7c-5f4f-41db-9645-9e8d97f94e3e","resolution":{"observed_at":"2026-08-04T16:17:17.549676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:15.089461Z","title":"For BabyHuBERT-1, we extract features from the 6th layer of WavLM-base-plus","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:15.089461Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:bc5dbd6603a0564a06d26ef508f88c88b9f91aefb7d5eb94b507de7abcb1032d","observation_id":"48f06c87-4cdd-4cb6-81b1-981995ff4aaa","resolution":{"observed_at":"2026-08-04T16:17:15.089461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.766857Z","title":"An open-source voice type classifier for child-centered daylong recordings,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.766857Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:a7cbcd2cdbe25ab85f4ce8cb6d11798c62534d1576d1c449e68c12bf5ab6f604","observation_id":"c04219f6-2123-4d13-b872-af115a88c0bb","resolution":{"observed_at":"2026-08-04T16:17:16.766857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:16.859633Z","title":"Signal processing for young child speech language development.,","venue":null,"work_id":null,"year":2008},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:16.859633Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:971cddae7bbe6070c334632af26ffbe60cab586c688ae242fe0cce3bf1885f4e","observation_id":"22b0a1c0-bf6b-4fda-91a7-6d7c4f6bed64","resolution":{"observed_at":"2026-08-04T16:17:16.859633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.009526Z","title":"Speaker recognition from raw waveform with sincnet,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.009526Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:6b7db4ed39c56223dd218112072714015c744586efe90ef191688cb1e399918c","observation_id":"bf2eca15-1ff9-4885-ba3a-68295aecf0e3","resolution":{"observed_at":"2026-08-04T16:17:17.009526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.129113Z","title":"Challenges in Automated Processing of Speech from Child Wearables: The Case of V oice Type Classifier,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.129113Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:069af7e38e4908cd4e040a3591654a67ca459a4a414e259d7ba417a1610843dd","observation_id":"65504122-2325-4306-89d8-5b34d2807105","resolution":{"observed_at":"2026-08-04T16:17:17.129113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.275963Z","title":"Developing a cross-cultural annotation system and metacorpus for studying infants’ real world language experience,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.275963Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:9869412e92e820af6a56a699a493a7a219fbd10975036da03ea7b28e651ae0f3","observation_id":"e5ff6c42-0ded-4034-9da0-f17e535428c1","resolution":{"observed_at":"2026-08-04T16:17:17.275963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.359917Z","title":"Wavlm: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.359917Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:8e2a3799aa6ba275d4b7fe276f11332a2b94ce6828dc7e0b895b6023d55cc608","observation_id":"55f4bc48-c8bd-4414-895d-46353011723e","resolution":{"observed_at":"2026-08-04T16:17:17.359917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.785559Z","title":"Hubert: Self-supervised speech representation learning by masked prediction of hid- den units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.785559Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:c44391fd63d15a50d45e10eafe727c29f7014a5fb350ef25d3f101f74f3557c7","observation_id":"16dfa19a-e731-410e-bb84-cf5f3b0f51f1","resolution":{"observed_at":"2026-08-04T16:17:17.785559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:17.961650Z","title":"Superb: Speech processing universal performance benchmark,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:17.961650Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:6b3c2ef01f43573f2d74efcfe714e3646661e8c55e113b1fd6d11f99cfee8220","observation_id":"cfe825c6-7bba-418e-8b32-d845b1e1b0be","resolution":{"observed_at":"2026-08-04T16:17:17.961650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.159359Z","title":"pyannote. metrics: A toolkit for reproducible evaluation, diagnostic, and error analysis of speaker diarization systems.,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.159359Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:779f7b4be8ead54d104eb0c82c225df054ee16ab547284bb5ca3d32781971743","observation_id":"dfe15651-fce5-4a50-9d6b-31680f076f1f","resolution":{"observed_at":"2026-08-04T16:17:18.159359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.216118Z","title":"Torchaudio 2.1: Advancing speech recognition, self-supervised learning, and audio process- ing components for pytorch,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.216118Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:85385be0ae942b4e05685536ca6d9bccb21494c1ea26b91ef31a5e8f5782974b","observation_id":"7cdeb22e-d547-4360-b9d4-1469b816b4a0","resolution":{"observed_at":"2026-08-04T16:17:18.216118Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.319135Z","title":"Scikit-learn: Machine learning in Python,","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.319135Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:d29081bf508b2b7496a228f181969732de415bbda7aeede484d27458708548cf","observation_id":"5ac68a9b-1902-4b7b-8171-71b48a6242cb","resolution":{"observed_at":"2026-08-04T16:17:18.319135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.361461Z","title":"Child-directed and overheard input from different speakers in two distinct cultures,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.361461Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:28fbc799a359a84f5a69b7e524e9659fe25bb8068fa71c1d3c914202d7d4ea70","observation_id":"6e801bbe-5292-40a0-b279-9a4ff93a2c25","resolution":{"observed_at":"2026-08-04T16:17:18.361461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T16:17:18.443046Z","title":"Putting the child in the driver’s seat: insights into language development from children’s interactions in preschool classrooms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T16:17:18.443046Z"},"links":{"citing_paper":"/paper/2509.15001"},"observation_digest":"sha256:a932ba7ddef36960b7120a8c5002fed40b4aea00ea654653c61a8b87b6e0e7f5","observation_id":"f6fc1119-c1c3-47ca-9202-f2741eb1efff","resolution":{"observed_at":"2026-08-04T16:17:18.443046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.15001","last_updated":"2026-06-29T13:06:46Z","latest_version":3,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-13T07:04:49.088347Z","submitted_at":"2025-09-18T14:34:17Z","title":"BabyHuBERT: Multilingual Self-Supervised Learning for Segmenting Speakers in Child-Centered Long-Form Recordings"},"reference_resolution":{"displayed":35,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":34,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":35},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 35 of 35 outbound references and 4 inbound Pith citation observations for arXiv:2509.15001."}