{"as_of":"2026-08-11T19:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:683bf0745af74bf2db9e06aa05b082f0f2f26a34251325f7d83239925276f74c","coverage":[{"denominator":35,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":35,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T22:27:51.363256Z","state":"measured"},{"denominator":35,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":35,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-11T06:34:44.6726+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2501.01674/citation-record","integrity":"/paper/2501.01674/integrity","json":"/paper/2501.01674/citation-record.json","paper":"/paper/2501.01674"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.814572Z","title":"Heterogeneous face recognition via face synthesis with identity- JOURNAL OF LATEX CLASS FILES, VOL. 14, NO. 8, AUGUST 2015 5 attribute disentanglement,","venue":null,"work_id":"c6cbf5a3-49e4-4607-a5be-33fe2097c9f2","year":2015},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.213344Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:112c069ea603cc446cc937059bbadcd2b379a03e3acfccfc4e642b7f33e493ab","observation_id":"f8ba8118-8bd6-48c6-86c4-b529681912d3","resolution":{"observed_at":"2026-08-10T22:27:51.818944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.803326Z","title":"Cont rollable and guided face synthesis for unconstrained face recognition,","venue":null,"work_id":"51efdc36-a902-4b0a-936b-d94b55b9b5c4","year":2022},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.218340Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:bf629ba3b21ab2e91f927328bb343c68e8f40904fdadd5247611c1f0095b212c","observation_id":"29a387b7-caeb-4855-aa8b-c5e1abbfd49b","resolution":{"observed_at":"2026-08-10T22:27:51.807118Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.791470Z","title":"Privacy-preserving a nnotation of face images through attribute-preserving face synthesis,","venue":null,"work_id":"548155b3-44f9-43c7-a5e2-f2a4988f4a34","year":2019},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.222391Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:9de95b5f075e49fdb4ab96f49cd58f789450069d796707ff948cd8df56a44502","observation_id":"297108f3-2331-4151-be08-c4d9fa53821d","resolution":{"observed_at":"2026-08-10T22:27:51.795759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.780307Z","title":"Is synthetic data all we nee d? benchmarking the robustness of models trained with synthet ic images,","venue":null,"work_id":"0c0119c5-6362-4d53-ac3c-fdd023e61475","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.227040Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:9aad03cdfc3df4d47f617a8219cf0310a08628181848a3eae3db82dcbbe65721","observation_id":"83f2912c-8868-4599-8e4f-1536112649f7","resolution":{"observed_at":"2026-08-10T22:27:51.784171Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.768474Z","title":"Attgan: Facial attribute editing by only changing wh at you want,","venue":null,"work_id":"42d09e2a-732b-41c1-990c-4206bed34178","year":2019},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.231188Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:1642b382873e8c3f4b03805d1c25b7363cecb478994d95ba9e096da6d2c4554d","observation_id":"bd6e9c1d-82b3-400b-8157-9064c0af1dbe","resolution":{"observed_at":"2026-08-10T22:27:51.772233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.756986Z","title":"Guidedstyle: Attribute knowledge guided style manipulation for semantic face editing,","venue":null,"work_id":"cbfd02bb-30ba-4d2b-8564-479212925351","year":2022},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.235482Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:cbad0d5e023457355d5e6904a5b1000d87e2988ed9f8fefcf608834ddb072ee1","observation_id":"d1919790-1467-4ec5-ab94-c6460c6ba212","resolution":{"observed_at":"2026-08-10T22:27:51.760951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.745828Z","title":"Icgnet: An intensity-controllable generation ne twork based on covering learning for face attribute synthesis,","venue":null,"work_id":"8862d709-88ed-4fd7-9689-fecebe933e8d","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.240146Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:591f24e767cf1d39142d967b3d4a5994a39463a2d5a4a06b988fe2bc18b78f5d","observation_id":"009b8049-929f-4133-ab60-8157ca10c834","resolution":{"observed_at":"2026-08-10T22:27:51.749806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.734226Z","title":"Speech ﬂuency: effe ct of age, gender and context,","venue":null,"work_id":"57bd201b-1295-4bd6-98bd-ddfc5d890afa","year":1995},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.244331Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:9af55de518befd3121fd73799f70aa037c7f7e75345d43fafed858e9cbbf81e2","observation_id":"3d70aadb-db06-4cc7-99d9-6b8090e3c098","resolution":{"observed_at":"2026-08-10T22:27:51.738579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.722480Z","title":"Ag e- related effects on speech production: A review,","venue":null,"work_id":"24f21bbb-399d-4dae-9ae0-c5269c7b8de2","year":2006},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.249125Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:e83e97e80965afa350b07dbad49a71cccb7a96daeb68966dc9fb22197763fd19","observation_id":"b9f341dc-22a3-4d7e-9482-f9e8fdf85a16","resolution":{"observed_at":"2026-08-10T22:27:51.726635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.710735Z","title":"A novel method for classifying body mass index on the basis of s peech signals for future clinical applications: a pilot study,","venue":null,"work_id":"52bc2c9f-074f-40e0-b3c4-0c9e7e41bed2","year":2013},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.253797Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:cf5df3396733525f994297574d10e0fdf7aab8ad59c0d3b473454449672ce7e7","observation_id":"3488a3fe-ee55-4abb-945d-d9ea62783193","resolution":{"observed_at":"2026-08-10T22:27:51.715019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16104","last_updated":"2024-04-24T18:00:06Z","snapshot_observed_at":"2026-08-10T15:22:06.452945Z","submitted_at":"2024-04-24T18:00:06Z","title":"Evolution of Voices in French Audiovisual Media Across Genders and Age in a Diachronic Perspective","version":1},"cited_work":{"arxiv_id":"2404.16104","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.16104","snapshot_observed_at":"2026-08-10T22:27:51.488494Z","title":"Evolution of Voices in French Audiovisual Media Across Genders and Age in a Diachronic Perspective","venue":"eess.AS","work_id":"c31f4d5b-b3e5-4efc-8508-324227fdeaf2","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.257722Z"},"links":{"cited_paper":"/paper/2404.16104","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:18e4afc9e8dade100342bab04e616f5f02a111c7b7fd118a988b23b3ce566f56","observation_id":"8d43cd9f-9d7b-4567-9b34-b4fd0ed5f6a5","resolution":{"observed_at":"2026-08-10T22:27:51.492868Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.698583Z","title":"Learning utterance-l evel represen- tations for speech emotion and age/gender recognition usin g deep neural networks,","venue":null,"work_id":"efb62a2c-b475-4652-b531-0350fa71be83","year":2017},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.263882Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:a7815c33ea95daa0c02353fd11e7e38fbe820d05d61fce96b463aac0d8c89ef2","observation_id":"72df8c64-0071-4b53-ac07-cbffc7f4dd01","resolution":{"observed_at":"2026-08-10T22:27:51.703208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.685217Z","title":"Age and gender recognition using a convolutional neural ne twork with a specially designed multi-attention module through speec h spectro- grams,","venue":null,"work_id":"cc8281e3-e046-42fc-8aa1-e6e7b6816f4b","year":2021},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.267904Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:96457f03a0fb0accfc6505adb7afe8d414b18786f7e43621b08ebc09674cc21a","observation_id":"2381a2a5-98ec-4655-b2ba-a38008fd2bed","resolution":{"observed_at":"2026-08-10T22:27:51.689606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.673532Z","title":"Speech-based age and gender predicti on with transformers,","venue":null,"work_id":"f57298fa-78cb-49e1-a1da-666bfe7d2553","year":2023},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.271754Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:af7899304d1ebf0a102328785ffd2d0a3a28b4fa0e522cb22a509319a7cbe03e","observation_id":"a7049c80-06a8-4a7c-9615-e3a291feab9b","resolution":{"observed_at":"2026-08-10T22:27:51.677494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.661334Z","title":"Privacy-oriented manipulation of speaker representatio ns,","venue":null,"work_id":"0fb5b711-6528-4994-aaf9-eca8c5fc011f","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.276063Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:c2efd599d21d4b8fd4022a535d1782eb6012249d63bdf11a8dc62a4e721f0eb8","observation_id":"926b4e5d-a9f5-4c93-b2d0-c52eced30a25","resolution":{"observed_at":"2026-08-10T22:27:51.665429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.649879Z","title":"Investigating the contribution of speaker attributes to speaker separability using disent angled speaker representations,","venue":null,"work_id":"bc705dfd-0f6d-42f2-8661-b781395d9b12","year":2022},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.280246Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:c37b7ae6ca57248b074ca7c4a50ef609c372f21169ef5bfe8ba15e376598cab1","observation_id":"9d5bbfe2-d96f-4fd6-a0ca-0091a6398bf2","resolution":{"observed_at":"2026-08-10T22:27:51.653807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.637741Z","title":"Adversarial-fr ee speaker identity-invariant representation learning for automati c dysarthric speech classiﬁcation.,","venue":null,"work_id":"848fb42c-e48d-4cd9-b86c-c7254088393d","year":2022},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.285169Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:d1c3b2aa80722c9b057e1262ace2e2592313eaeb83178c0fec781e09f9864108","observation_id":"15871c13-973d-4f14-ba24-3fa26ba79dea","resolution":{"observed_at":"2026-08-10T22:27:51.642081Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12399","last_updated":"2025-03-27T13:14:57Z","snapshot_observed_at":"2026-08-02T15:16:01.921143Z","submitted_at":"2024-10-16T09:27:25Z","title":"SF-Speech: Straightened Flow for Zero-Shot Voice Clone","version":2},"cited_work":{"arxiv_id":"2410.12399","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.12399","snapshot_observed_at":"2026-08-10T22:27:51.472526Z","title":"SF-Speech: Straightened Flow for Zero-Shot Voice Clone","venue":"cs.SD","work_id":"174a98ad-ac3e-4928-bf0d-4d7a4aa1f978","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.289783Z"},"links":{"cited_paper":"/paper/2410.12399","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:89a40fef99c8dd046c693455cdbbbee617dd3008ce6026a0d2b4460cabaac60e","observation_id":"9931ffb5-5bba-4518-b988-fa8f4ebac79c","resolution":{"observed_at":"2026-08-10T22:27:51.476746Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05407","last_updated":"2024-07-09T07:42:51Z","snapshot_observed_at":"2026-08-10T17:49:50.848957Z","submitted_at":"2024-07-07T15:16:19Z","title":"CosyVoice: A Scalable Multilingual Zero-shot Text-to-speech Synthesizer based on Supervised Semantic Tokens","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05407","snapshot_observed_at":"2026-08-10T22:27:51.294397Z","title":"Cosyvoice : A scalable multilingual zero-shot text-to-speech synthes izer based on supervised semantic tokens,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.294397Z"},"links":{"cited_paper":"/paper/2407.05407","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:64b960c597a8ae1274af6f59b621752d6e8f58d575b2caa2933036f77115de25","observation_id":"6aa991c2-02e0-4334-b3b7-742657367a55","resolution":{"observed_at":"2026-08-10T22:27:51.294397Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02430","last_updated":"2024-06-04T15:48:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-04T15:48:29Z","title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02430","snapshot_observed_at":"2026-08-10T22:27:51.299207Z","title":"Seed-tts: A family of high-quality versatile speech gener ation models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.299207Z"},"links":{"cited_paper":"/paper/2406.02430","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:0ccbb15ec4ba1825ab17fb2e17889522e4415653d542eefb9f5883a9dd9c64a0","observation_id":"43ba1f5f-e216-4ff4-889e-5c742f0375ca","resolution":{"observed_at":"2026-08-10T22:27:51.299207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05608","last_updated":"2025-03-27T06:27:57Z","snapshot_observed_at":"2026-08-04T04:39:26.032931Z","submitted_at":"2024-07-08T04:48:43Z","title":"A Benchmark for Multi-speaker Anonymization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05608","snapshot_observed_at":"2026-08-10T22:27:51.304619Z","title":"A b enchmark for multi-speaker anonymization,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.304619Z"},"links":{"cited_paper":"/paper/2407.05608","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:d3c0b3c3389b909a23bbe25fdf34c0a2bbcefcc658eba233329314254fb35b75","observation_id":"430bef79-2046-4417-951e-baa5b92d63e4","resolution":{"observed_at":"2026-08-10T22:27:51.304619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.624475Z","title":"Speaker anonymization using neural audio codec language models,","venue":null,"work_id":"8e45873e-e1e1-4c60-af88-efe5b16f352c","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.309041Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:983e83bbc3607cad2024bd2b869b27d0179d35769a4aa3d07c7418c4b9536aea","observation_id":"9d0fa682-1bce-4f30-ae32-0b70d0907065","resolution":{"observed_at":"2026-08-10T22:27:51.628986Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.609639Z","title":"Distinctive and natural speaker anonymization via singul ar value transformation-assisted matrix,","venue":null,"work_id":"30c76496-4cc1-4f5b-8a72-451b2119aaeb","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.314127Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:0761cbf515b97053051e5011a67638296c27fe2909b791d1e04102958445435e","observation_id":"7b5d9f65-e4d3-4137-98a6-f4b3b2b132e8","resolution":{"observed_at":"2026-08-10T22:27:51.614074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.597336Z","title":"Synthe-sees: Face based text-to-speech for virtual speak er,","venue":null,"work_id":"68a0a54f-23db-4529-b1f5-391b8372c69e","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.318287Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:92a996c2e9a90784778fc2d810776ba51f7ff44a671f242636a0d44e85a0aeac","observation_id":"086be671-bdb6-4b08-aa24-877b8c379de9","resolution":{"observed_at":"2026-08-10T22:27:51.601798Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.584507Z","title":"Fvtts: Fac e based voice synthesis for text-to-speech,","venue":null,"work_id":"8376922b-a159-4e8e-8172-21fa1fd0661c","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.321970Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:8281a988690f6befa8f902ff392e3d1583375fa69249a40fd9a2fc108e903fa0","observation_id":"66a64e5e-1755-4c17-b4cb-faee84ed48eb","resolution":{"observed_at":"2026-08-10T22:27:51.588586Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16314","last_updated":"2024-06-24T04:46:50Z","snapshot_observed_at":"2026-07-06T18:35:49.991098Z","submitted_at":"2024-06-24T04:46:50Z","title":"DreamVoice: Text-Guided Voice Conversion","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.16314","snapshot_observed_at":"2026-08-10T22:27:51.326190Z","title":"Dreamvoice: Text-guided voice conversion,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.326190Z"},"links":{"cited_paper":"/paper/2406.16314","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:8f596f1656d035ef7df9f41a662603e759defbec5bf70bc838dc48a69af6ab24","observation_id":"13d1ae39-34e6-4e71-a0dc-4e35fc8789d5","resolution":{"observed_at":"2026-08-10T22:27:51.326190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08812","last_updated":"2024-06-13T05:06:30Z","snapshot_observed_at":"2026-07-06T18:30:06.570612Z","submitted_at":"2024-06-13T05:06:30Z","title":"Generating Speakers by Prompting Listener Impressions for Pre-trained Multi-Speaker Text-to-Speech Systems","version":1},"cited_work":{"arxiv_id":"2406.08812","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.08812","snapshot_observed_at":"2026-08-10T22:27:51.408256Z","title":"Generating Speakers by Prompting Listener Impressions for Pre-trained Multi-Speaker Text-to-Speech Systems","venue":"cs.SD","work_id":"df9db7e7-0d40-468b-a796-cc86cce5e19b","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.330570Z"},"links":{"cited_paper":"/paper/2406.08812","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:8a4dd000f009dd33aa2b0217fe3fc9a8662a81f96e3dde22fa6de3195050e769","observation_id":"0ca9e045-e9e6-4e27-8692-cf9f107485c0","resolution":{"observed_at":"2026-08-10T22:27:51.415034Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.571740Z","title":"Fine-grained and interpretable neural speech editing,","venue":null,"work_id":"0d386722-c168-41cf-8203-60fbba6b0a0c","year":2024},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.335171Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:9dda3f892f10d5c8753d272e1da53769c973ef87f7e4e862869653c6ace928e0","observation_id":"68a6cc44-0d69-4972-b84d-f1b3d99c3552","resolution":{"observed_at":"2026-08-10T22:27:51.576367Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.339675Z","title":"Wavlm: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.339675Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:efe1026669ec4a5b73a66585e38ad0116a05b530ee8c06cc9482768f129e8a16","observation_id":"f8c8973c-6f72-4f18-8251-b5767829bb32","resolution":{"observed_at":"2026-08-10T22:27:51.339675Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.552121Z","title":"Disen tangled information bottleneck,","venue":null,"work_id":"267049d5-61ba-4af5-bbe0-867a0131433b","year":2021},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.343757Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:bb1ca68208a72ab69d7fcd38e45ed21b1b260e62f525a745546ab9732825f4bf","observation_id":"f2e99877-48ee-4535-8b0b-4ac5123b3138","resolution":{"observed_at":"2026-08-10T22:27:51.556216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.539335Z","title":"Arbitrary style transfe r in real-time with adaptive instance normalization,","venue":null,"work_id":"ce98b587-a23c-4e3b-bbc3-1a53274f51d3","year":2017},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.347492Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:4cbd55d81c55e261df538cda12dfb869171bb41fc2de54ff17d2039b65929d32","observation_id":"710bcab5-5a28-4239-b82d-ce6b39e85402","resolution":{"observed_at":"2026-08-10T22:27:51.543731Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.03003","last_updated":"2022-09-07T08:59:55Z","snapshot_observed_at":"2026-07-06T13:49:40.974495Z","submitted_at":"2022-09-07T08:59:55Z","title":"Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.03003","snapshot_observed_at":"2026-08-10T22:27:51.351347Z","title":"Flow strai ght and fast: Learning to generate and transfer data with rectiﬁed ﬂ ow,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.351347Z"},"links":{"cited_paper":"/paper/2209.03003","citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:2412c5eb9e79547aaab6e90634cf5ec17e41de432cc98a9032153d030ab3888d","observation_id":"0c48df15-a763-4d55-8ad4-c7169f0b8ad9","resolution":{"observed_at":"2026-08-10T22:27:51.351347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.526908Z","title":"V oxceleb2: Deep speaker recognition,","venue":null,"work_id":"2eaccf31-9506-44e8-aa1f-7fa07d6fabfc","year":2018},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.355597Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:e61d0ed8b10473d8dc31c17f6289031e2b7ed69d5cf7720ce45ec9ad4a46e66e","observation_id":"a71467c8-5acf-41dd-a0d4-7daf97037a79","resolution":{"observed_at":"2026-08-10T22:27:51.530951Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.514068Z","title":"Age-vox-celeb: Multi-modal corpus for facial a nd speech estimation,","venue":null,"work_id":"050107ee-677f-40ec-919c-a268027bdd8b","year":2021},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.359540Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:6d6eedcab953cb2fc34bccd3c721c715da70921e74a89a1f579a13ea5402a7e0","observation_id":"905e03ee-d3a7-4488-b6cc-ed3d79e7bdf0","resolution":{"observed_at":"2026-08-10T22:27:51.518372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T22:27:51.501077Z","title":"Hiﬁ-gan : Generative adversarial networks for efﬁcient and high ﬁdelity speech s ynthesis,","venue":null,"work_id":"7ee55687-1216-44a5-a53b-7a34dedc03d8","year":2020},"citing_paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-10T22:27:51.363256Z"},"links":{"citing_paper":"/paper/2501.01674"},"observation_digest":"sha256:1cf57b8e7eee5640839470b045ca7239a08f9e4271f8c9e9a1f49093f5c573e6","observation_id":"d2a23e6b-0a35-459e-8e37-cd2d4e2fd538","resolution":{"observed_at":"2026-08-10T22:27:51.505639Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-11T06:34:44.6726+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2501.01674","last_updated":"2025-01-03T07:35:08Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-11T16:12:27.855365Z","submitted_at":"2025-01-03T07:35:08Z","title":"Controlling your Attributes in Voice"},"reference_resolution":{"displayed":35,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":6,"verified_exact":3,"verified_fuzzy":26},"total_outbound_references":35},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-11T06:34:44.6726+00:00","source":"crossref"},{"observed_at":"2026-08-11T06:34:36.301508+00:00","source":"retraction_watch"}],"thesis":"As of 11 August 2026, this Paper Citation Record lists 35 of 35 outbound references and 0 inbound Pith citation observations for arXiv:2501.01674."}