{"as_of":"2026-08-09T23:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:04f14ac4b92400de3e1f170640d59109bbab639fa07edf10cce991ad2bc23add","coverage":[{"denominator":73,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":73,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T11:08:24.855494Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2502.08669/citation-record","integrity":"/paper/2502.08669/integrity","json":"/paper/2502.08669/citation-record.json","paper":"/paper/2502.08669"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:28.428204Z","title":"A guide to deep learning in healthcare","venue":null,"work_id":"97556abb-73b2-4837-a958-50ee980bd005","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.772934Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7c955a541708ea17c719acf39a844c0f02fbe94e011f4de18e78285f4c5dfd62","observation_id":"752285f4-0db4-4373-9c17-ad9284c7da41","resolution":{"observed_at":"2026-08-08T11:08:28.541963Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:28.082487Z","title":"Mining electronic health records: towards better research applications and clinical care","venue":null,"work_id":"11cb66ff-15c7-49e0-bc83-b6ba8d19c641","year":2012},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.827138Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:6aa9b982ab1686f9a2b9952c02608e457ce2a69a6e02e01059f4434c342585fc","observation_id":"e500938e-9cd6-47b5-a89d-5cc84814eef3","resolution":{"observed_at":"2026-08-08T11:08:28.255848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.847394Z","title":"The Secondary Use of Electronic Health Records for Data Mining: Data Characteristics and Challenges","venue":null,"work_id":"dfe190d2-c009-426e-a183-8d0ec92afece","year":2022},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.947591Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:2e7bad2134e64de7c41bd6b484095867099b1c058dbcc02bfa44258204aab4bf","observation_id":"86a1d082-57e8-4ed7-ac6e-65df1a29e51d","resolution":{"observed_at":"2026-08-08T11:08:27.968411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.738087Z","title":"Mining Electronic Health Records (EHRs): A Survey","venue":null,"work_id":"606ba9a6-9d20-4e8c-901c-d87fec502023","year":2018},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.952192Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:b6d96476a6675595e91181b353c3744278bc61ecfeb32d3ed9fe6dc20baa5786","observation_id":"91bc46bf-d329-4ff9-9133-0d408cb38391","resolution":{"observed_at":"2026-08-08T11:08:27.790068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.723692Z","title":"Impact of Different Approaches to Preparing Notes for Analysis With Natural Language Processing on the Performance of Prediction Models in Intensive Care","venue":null,"work_id":"7b3a0b14-b932-426f-b11c-53c4f89a9f05","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.955930Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:ab712ff83f350854328462d445d2c82df601eb71aa8b63a8b38ac1606d2265eb","observation_id":"f1bba4c2-c4d1-4da0-a356-df10a68f68f9","resolution":{"observed_at":"2026-08-08T11:08:27.727668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.712052Z","title":"An advanced review on text mining in medicine","venue":null,"work_id":"fa901a6a-7beb-42d2-a7e4-5a99392cf1e7","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.959823Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:dd14d28db9bf58f3e7352e7e0e4b2178b42063ee24d236b473ea54ef9cd850ae","observation_id":"84acccdd-8073-42fc-ae5d-596b963c3f10","resolution":{"observed_at":"2026-08-08T11:08:27.716168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.702258Z","title":"Natural Language Processing for EHR-Based Computational Phenotyping","venue":null,"work_id":"39b3306f-11af-4937-905d-895ac45cde03","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.963686Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:451d82e25633b2f7fc91db04605f134ddfb95d79e580f0c9b5fc630c11f0e436","observation_id":"b9b9a6ad-45a7-414f-9056-b629527816ca","resolution":{"observed_at":"2026-08-08T11:08:27.705651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.692299Z","title":"Clinical Text Data in Machine Learning: Systematic Review","venue":null,"work_id":"dc0aa80c-fb2c-4a55-9c68-a44aa07b4062","year":2020},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.968244Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:8682dee29f4d54b21b8b374cfd7e222794aeb0867d33ee84b025da53bbe8979d","observation_id":"aca14b19-05bc-4b62-97aa-cbfb13ce3c02","resolution":{"observed_at":"2026-08-08T11:08:27.696140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.682137Z","title":"Attention is all you need","venue":null,"work_id":"e9499bed-65c0-405d-8e16-cb710a182c69","year":2017},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.971648Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7963545e933f2fc244637e8bbeb9e50f6fb7df6055e070e625089963292ce836","observation_id":"c5a585c0-6d5f-44e6-b287-eefa4c808f4e","resolution":{"observed_at":"2026-08-08T11:08:27.685647Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.672558Z","title":"Transformers for Multi-label Classification of Medical Text: An Empirical Comparison","venue":null,"work_id":"a41608c4-6348-4e65-bbcc-7f6a6c676f06","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.975890Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:54d2c18bdbb3a61eb35255bc282021c018bfd51224330b7c26b12db3462a15cc","observation_id":"a941e1f0-6e09-4a7e-8c9e-a37c23e2f332","resolution":{"observed_at":"2026-08-08T11:08:27.675972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.663402Z","title":"Measurement of Semantic Textual Similarity in Clinical Texts: Comparison of Transformer-Based Models","venue":null,"work_id":"6feaaba8-7d5f-40b3-9d7c-d94bfe5c60a8","year":2020},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.978899Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:ec41fb44406cdc0c529fa7cc4566e0f8c5e2ffe12cefb69612988e685543405c","observation_id":"f68e9968-f894-4003-8da7-0fcca30bcf6b","resolution":{"observed_at":"2026-08-08T11:08:27.666626Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.650978Z","title":"Limitations of Transformers on Clinical Text Classification","venue":null,"work_id":"6b00191f-cda8-4578-9769-195ae9ffbe2a","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.982495Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:f68d9b4ea376ef49d0095f78d020e94e9a536ab5555fa3d31b83ba63dcbf5549","observation_id":"8d142594-b7f7-4b9e-9140-616f04c8df3e","resolution":{"observed_at":"2026-08-08T11:08:27.654522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.640894Z","title":"A Multimodal Transformer: Fusing Clinical Notes with Structured EHR Data for Interpretable In-Hospital Mortality Prediction","venue":null,"work_id":"f2cf177a-83ff-479f-8a44-b90c8ff12c3c","year":2023},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.985535Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:45435e131f33da5df4d7302535734483c666f3a7b47d241d136b1da46136b86d","observation_id":"7ac4431c-e1ba-4de6-a4fc-23ce66dfb813","resolution":{"observed_at":"2026-08-08T11:08:27.644634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.18223","last_updated":"2026-03-18T05:34:39Z","snapshot_observed_at":"2026-08-06T23:27:24.356320Z","submitted_at":"2023-03-31T17:28:46Z","title":"A Survey of Large Language Models","version":19},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.18223","snapshot_observed_at":"2026-08-08T11:08:23.989528Z","title":"A Survey of Large Language Models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.989528Z"},"links":{"cited_paper":"/paper/2303.18223","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:759c9315060663bcb758446ea6cbf006de4137f0cdf28bcdee749ca3502341d2","observation_id":"918e2a77-27b1-40a0-8c0f-77ba5c6bc14d","resolution":{"observed_at":"2026-08-08T11:08:23.989528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.629586Z","title":"A Survey on Evaluation of Large Language Models","venue":null,"work_id":"3f8405e1-e5f4-43c7-af05-87281cb9657c","year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.993915Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:189eeb444c49a2aafc976e6be3c7004ee701e51eadec30fdbfa67e509908cc29","observation_id":"ee349625-56e4-4ec8-8131-300d76dbc2f1","resolution":{"observed_at":"2026-08-08T11:08:27.633791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:23.997532Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:23.997532Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:6da30ce904c18850c5e1ac88b6bd16939e76023483daed20bba5b533c5c4b861","observation_id":"f21d76a8-2314-4c5d-a99a-16258dc9f593","resolution":{"observed_at":"2026-08-08T11:08:23.997532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06196","last_updated":"2025-03-23T14:51:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-09T05:37:09Z","title":"Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.06196","snapshot_observed_at":"2026-08-08T11:08:24.002580Z","title":"Large Language Models: A Survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.002580Z"},"links":{"cited_paper":"/paper/2402.06196","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:3d0f5eea102fc4f729e803a32ec663e0bdd9dd4a39c7abaef0c1b26962968dfe","observation_id":"02b08d9d-3170-4c97-9e09-c1e92556a213","resolution":{"observed_at":"2026-08-08T11:08:24.002580Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:24.007687Z","title":"Attention is not all you need: the complicated case of ethically using large language models in healthcare and medicine","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.007687Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:2ecec086ca258f87b93515358eb819c0608bf89a309d0a74cc40b396e83388df","observation_id":"2a28260a-ffd4-4723-acb0-645789180932","resolution":{"observed_at":"2026-08-08T11:08:24.007687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.562713Z","title":"BioBERT: a pre-trained biomedical language representation model for biomedical text mining","venue":null,"work_id":"bbfe45b0-01e5-4269-8828-f6f1b2741238","year":2020},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.011979Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:a37ce5b0dd44e891896f8b91f29647934b8cd1bfead255a7205bbb5f00f84201","observation_id":"e3b6ed23-b200-4a46-8bc0-8717fe2a0939","resolution":{"observed_at":"2026-08-08T11:08:27.623291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1904.05342","last_updated":"2020-11-29T03:40:45Z","snapshot_observed_at":"2026-07-06T07:45:18.053726Z","submitted_at":"2019-04-10T17:53:13Z","title":"ClinicalBERT: Modeling Clinical Notes and Predicting Hospital Readmission","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1904.05342","snapshot_observed_at":"2026-08-08T11:08:24.015849Z","title":"ClinicalBERT: Modeling Clinical Notes and Predicting Hospital Readmission","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.015849Z"},"links":{"cited_paper":"/paper/1904.05342","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7ecb7be3e90dbe55ff9a7710e53574a43b6e3cf2eb9a4d57e8b5f48d77c3854a","observation_id":"ecf920fc-2ef3-491a-b716-c382345424e7","resolution":{"observed_at":"2026-08-08T11:08:24.015849Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.335000Z","title":"Domain-Specific Language Model Pretraining for Biomedical Natural Language Processing","venue":null,"work_id":"9498f511-7480-4e29-ad9b-54b2bd056a71","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.020181Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:95ebb7929a4d6722e814d88509fdcf01523fbb7ac19c08b3cfc4609b40364517","observation_id":"4d9f88e0-dce1-40ef-a96f-f608cd12ca0f","resolution":{"observed_at":"2026-08-08T11:08:27.453357Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.144099Z","title":"Don’t Stop Pretraining: Adapt Language Models to Domains and Tasks","venue":null,"work_id":"7c72f158-eeda-4f93-8ae3-99bee84d36f7","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.024752Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:ccf53397c968c64b1220b2a7719f484622349c6a8de1bdcce540997155a27bc3","observation_id":"b4b54c9a-7d61-47fe-85ac-a18784b36b16","resolution":{"observed_at":"2026-08-08T11:08:27.226089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.026099Z","title":"SciBERT: A Pretrained Language Model for Scientific Text","venue":null,"work_id":"49c959a1-5c0f-4b73-b2d6-b0a50285c2cd","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.029851Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:05cabeb833b2ae47fb65116b176a57752779489d476aa56384230c4f1b89eada","observation_id":"2bfa317a-d407-4b7b-b4ce-34cf53328d38","resolution":{"observed_at":"2026-08-08T11:08:27.029609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.015274Z","title":"MedCPT: Contrastive Pre-trained Transformers with large-scale PubMed search logs for zero-shot biomedical information retrieval","venue":null,"work_id":"88cffd68-8f39-4c14-8ca7-bd528330114d","year":2023},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.034489Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:b896487ea7b4c1f55e12bfe421b033cb1d516dc4dc72902628e96a9104c4c6ee","observation_id":"2da7d34d-e9cb-40b0-bed5-09af605b6ea6","resolution":{"observed_at":"2026-08-08T11:08:27.019482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:27.004186Z","title":"A large language model for electronic health records","venue":null,"work_id":"1f7e1183-6f51-4fcd-8205-4f5ccf6332d6","year":2022},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.038881Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:06df33e7b5bbb1d16ca0599d16d244c616e3034b4db3f93e531ebd8c99f8aab4","observation_id":"ba3ffdeb-4ae6-46dd-8b23-f4165d76e7c6","resolution":{"observed_at":"2026-08-08T11:08:27.008132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.995245Z","title":"MIMIC-III, a freely accessible critical care database","venue":null,"work_id":"d57e6945-862f-4190-99a1-d1cb9f43e756","year":2016},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.042716Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:c356182bebea2ac1b48eafc227667a98cfe25343a19654d09b9fa4468c890a30","observation_id":"fde377b3-ab5c-4ac0-9f52-ed3ab168dfac","resolution":{"observed_at":"2026-08-08T11:08:26.998308Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.985276Z","title":"Annotating longitudinal clinical narratives for de- identification: The 2014 i2b2/UTHealth corpus","venue":null,"work_id":"5e41312b-5ee2-4c6c-9332-1a56164a5ebe","year":2014},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.046882Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:3f6a91b67705f9e02fc0882642c92dd169056d58a2ea70f42b4a55ca2bb0ac78","observation_id":"86ec5205-69b2-443f-948f-615d5ca2ef07","resolution":{"observed_at":"2026-08-08T11:08:26.988723Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.975604Z","title":"Automated systems for the de-identification of longitudinal clinical narratives: Overview of 2014 i2b2/UTHealth shared task Track 1","venue":null,"work_id":"6fa28e7b-a57b-4c88-9167-ce0df10e82d5","year":2014},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.073315Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:e31cb8b2ea22d07ae0876f0bed98068f2e8fe8a0fc28862661440ec79ed66139","observation_id":"39f02fb9-6958-4022-89a0-5ccdce5628bb","resolution":{"observed_at":"2026-08-08T11:08:26.979449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.964742Z","title":"Overview and Importance of Data Quality for Machine Learning Tasks","venue":null,"work_id":"1dc1c71d-aca1-43be-800b-403cf75498df","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.118799Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:a24987a8a6d51dff49e23430b37494e1f7efa178aed08fda6e97d94d6700e80f","observation_id":"3e87ac41-92f5-403e-beb8-baf12b9b095d","resolution":{"observed_at":"2026-08-08T11:08:26.968251Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.05935","last_updated":"2021-09-05T11:35:06Z","snapshot_observed_at":"2026-07-06T11:37:55.847817Z","submitted_at":"2021-08-12T19:22:27Z","title":"Data Quality Toolkit: Automatic assessment of data quality and remediation for machine learning datasets","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.05935","snapshot_observed_at":"2026-08-08T11:08:24.217434Z","title":"Data Quality Toolkit: Automatic assessment of data quality and remediation for machine learning datasets","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.217434Z"},"links":{"cited_paper":"/paper/2108.05935","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:495acb1445654b7fc6bcc278b2ad6108343f3788b61355d1cc2e71d82e037561","observation_id":"6eb22d96-ad66-434b-986f-e5bd8a9f0963","resolution":{"observed_at":"2026-08-08T11:08:24.217434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.954938Z","title":"Data Validation for Machine Learning","venue":null,"work_id":"edf400a5-27a8-4fba-bf08-9ea6c8020cb0","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.310600Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:a379455b473803ab5b6e80e973a85f453376a8186a3f1ab6bcd2d72804cd7135","observation_id":"febe3d98-1b05-42db-9021-8457e3de93e6","resolution":{"observed_at":"2026-08-08T11:08:26.958849Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.944583Z","title":"Data Evaluation and Enhancement for Quality Improvement of Machine Learning","venue":null,"work_id":"e63ad101-13ce-49d2-8a42-23fd1bebbae2","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.374451Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:425c40a13d11e145f8505870726f05cb843bdeae6bde445afafe937790596002","observation_id":"b712c440-d61c-4665-8166-9d36b0c69e41","resolution":{"observed_at":"2026-08-08T11:08:26.948188Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.800277Z","title":"Data quality considerations for big data and machine learning: Going beyond data cleaning and transformations","venue":null,"work_id":"6795d152-378d-466c-872b-a8c632af78b5","year":2017},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.440870Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:6a8475248c40aeeb63fb6b8094f7767b28dbfa03743e00f143a7e4b4b36de5f3","observation_id":"6e792519-8d6d-489b-be91-ccd94a0f123f","resolution":{"observed_at":"2026-08-08T11:08:26.889179Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.581777Z","title":"Data Readiness Report","venue":null,"work_id":"0f628397-b5b4-4538-bd01-81ff6d11c9cb","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.444703Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:89d86f804e152b719d3a43f55055992796cf9927ca107e97678633ac426c2b7e","observation_id":"0312a9fa-3ed0-4fcd-978e-df8f03957e7f","resolution":{"observed_at":"2026-08-08T11:08:26.710636Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.404721Z","title":"Executing Data Quality Projects Ten Steps to Quality Data and Trusted Information (TM)","venue":null,"work_id":"9fee055b-0839-4810-9163-78e30025355a","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.448916Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7150db68ea187b365948fee6ed0cb2bc302093153f124f8646dbdb8513abae77","observation_id":"e47f141f-d2c8-4afe-a6d2-d26c19755a2e","resolution":{"observed_at":"2026-08-08T11:08:26.526727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.372699Z","title":"A Short Review of the Literature on Automatic Data Quality","venue":null,"work_id":"b9a81026-5e01-4450-b31f-59fb14ea27ca","year":2022},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.452624Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:78396ddd570bba24bf4573e7e5c33a5266320a9e4a5478629ae71496ec9f1855","observation_id":"9342ac4e-67f7-4d69-8445-deea7da0dc58","resolution":{"observed_at":"2026-08-08T11:08:26.376779Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.359696Z","title":"A Practical Framework for Evaluating the Quality of Knowledge Graph","venue":null,"work_id":"c5220de0-13df-421e-9b43-4bc4c74e6fe2","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.456299Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:4b28ed52ce1fd4207a4f11b0c1004b71026c5ba4095bea6f3f5f94d3c871bf22","observation_id":"e8b9271f-3883-4e07-9d76-20c6814a8938","resolution":{"observed_at":"2026-08-08T11:08:26.364550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.349909Z","title":"Developing a systematic approach to assessing data quality in secondary use of clinical data based on intended use","venue":null,"work_id":"4023239b-18c5-407f-9f63-f7243bdb86c2","year":2022},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.459774Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:ed8a8931e5073ecbc5680806f55cf926bef5f8f87e42b7f61b1f36651850525c","observation_id":"222f17c8-d855-4616-8664-e48e275b6d46","resolution":{"observed_at":"2026-08-08T11:08:26.353256Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.337773Z","title":"Defining and measuring completeness of electronic health records for secondary use","venue":null,"work_id":"7d258a55-594c-46bf-bac9-57820315dfea","year":2013},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.462297Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:fd17a456608bdfbc7886a322031a365b0c265a3788493ef62bf1fa4960d43741","observation_id":"bcb03bce-30c8-409b-b18f-69a92050326d","resolution":{"observed_at":"2026-08-08T11:08:26.342194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.328332Z","title":"Review: Electronic Health Records and the Reliability and Validity of Quality Measures: A Review of the Literature","venue":null,"work_id":"bca298f4-1544-4f4a-8cfb-bab96bf5c9af","year":2010},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.465384Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:4d53277075389c6b37c4c836bb927973ce7de79278f0d7f8d750f9bbbd0204db","observation_id":"7f5ea6c2-a167-4e50-9e4c-d58f0d5eec13","resolution":{"observed_at":"2026-08-08T11:08:26.331447Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.318700Z","title":"Quality Indicators for Text Data","venue":null,"work_id":"0b4a1281-3897-405c-8596-deb19aa9006d","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.468632Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7ca5b96cb59723ce3ef420eb3d491179441cccd1ddbf7e2ebd2b0143960b1b42","observation_id":"994930ff-7ab6-4772-a9a4-44705a09f464","resolution":{"observed_at":"2026-08-08T11:08:26.322569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.307600Z","title":"Outlier Detection for Improved Data Quality and Diversity in Dialog Systems","venue":null,"work_id":"33ed9a5c-8d24-49a8-becb-3f26d0870f97","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.472034Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:f23080a8403409ace029a1dbb0da4aee9191f61bec936707b3418d334a096fa9","observation_id":"8f91165c-1ef7-4f35-ae8d-eb79948c2cd1","resolution":{"observed_at":"2026-08-08T11:08:26.311506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.297194Z","title":"Aiming beyond the Obvious: Identifying Non- Obvious Cases in Semantic Similarity Datasets","venue":null,"work_id":"1905e82a-2851-4665-9330-2b0093f10598","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.475327Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:a258c213f2b8d9d1346909ec6a7db5d5dfbee04d975a7b5ff3c1a2f7daa527e2","observation_id":"fd9f4e81-e76a-4096-bcc9-ec84422f6c80","resolution":{"observed_at":"2026-08-08T11:08:26.301072Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.285704Z","title":"Evolutionary Data Measures: Understanding the Difficulty of Text Classification Tasks","venue":null,"work_id":"c856f0c8-f51f-4733-bb68-9684bed3a39a","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.478749Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:194bb53de4831a62b15b00ffd25026c0c7440cf3b6d6534b47ddbc3ec4a73121","observation_id":"7d8b025e-6d17-4bb6-bf4f-7b532b38b50b","resolution":{"observed_at":"2026-08-08T11:08:26.290614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:26.074920Z","title":"Improving Neural Response Diversity with Frequency-Aware Cross-Entropy Loss","venue":null,"work_id":"da9e1a4c-a1ba-477a-b075-734b8e4c72f9","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.482451Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:405c426729e13d65c34723d95bad9c4f9d7f8c09ab26a82a571e8160b6dc44f9","observation_id":"081b4ace-b7f6-4eed-a936-88ab0d049938","resolution":{"observed_at":"2026-08-08T11:08:26.152677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.965465Z","title":"Confident Learning: Estimating Uncertainty in Dataset Labels","venue":null,"work_id":"78c7a821-a627-4482-ae2e-6f91fa5b0c24","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.485973Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:49953e7abde36beefa71d8604d380e88e1ba7d87d9398b50d38177e162ccfe73","observation_id":"e116d2fd-25c2-4b2b-8c47-d7c01f446b0a","resolution":{"observed_at":"2026-08-08T11:08:26.029486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.955474Z","title":"Detecting errors in part-of-speech annotation","venue":null,"work_id":"55d9f96e-68ef-4ed8-9250-4c82d861f752","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.488690Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:a8921d99fde37361cc8a46b8e65c7ff9fde23aff6cdb4e3b64199103b57acb0b","observation_id":"c538d4c7-9a84-4059-9e76-f7f5c5dabe9e","resolution":{"observed_at":"2026-08-08T11:08:25.958950Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.945444Z","title":"Detecting errors within a corpus using anomaly detection","venue":null,"work_id":"bd378e2e-3bb9-4b36-aeeb-8672b8ddd876","year":2000},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.492290Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:6a03d1bc0fd496830b62078633b13e8b64261e5acfa362c529ee01c21848446d","observation_id":"1db77e0a-b9bc-402d-9165-a62abcb09921","resolution":{"observed_at":"2026-08-08T11:08:25.949138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.934297Z","title":"Detecting annotation noise in automatically labelled data","venue":null,"work_id":"eb864cd5-5468-4070-8787-b094bdc3b118","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.496466Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:9d6a2efa6deb847eaa2c524b6bc6f3e6341969c7ea7406c771daf102c32cf6f1","observation_id":"d9f99329-ce3f-4594-b814-e114b2f1e035","resolution":{"observed_at":"2026-08-08T11:08:25.937736Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.922367Z","title":"Data Quality for Machine Learning Tasks","venue":null,"work_id":"576efea0-38fc-49a6-a401-a88517ecd0b9","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.500023Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:86168640164648df05e3a0edf1e0f6e560659ba9944f623d6539c22e0e10052f","observation_id":"a4c4469b-22f9-4848-92a3-58315b77dea9","resolution":{"observed_at":"2026-08-08T11:08:25.926965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.911106Z","title":"Dataset Cartography: Mapping and Diagnosing Datasets with Training Dynamics","venue":null,"work_id":"bfd8f09a-a3b2-4296-9fa3-fde679042779","year":2020},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.503533Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:43061e934464d03d8b8ddccdef576a5de6084809ded5add4f9f76c4311561273","observation_id":"f12bbf17-6848-46d2-8431-87a32a9590e4","resolution":{"observed_at":"2026-08-08T11:08:25.914995Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.900543Z","title":"Beyond Accuracy: Behavioral Testing of NLP Models with CheckList","venue":null,"work_id":"e8c8ab1f-2c78-4f2b-aafd-b66197b04269","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.507479Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:d2e326473a35355eec5b1705dafdd98a0ec018260013c3aa828a10b0247eb880","observation_id":"25322a57-536d-472f-85fc-fa691ca1a474","resolution":{"observed_at":"2026-08-08T11:08:25.904432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.890460Z","title":"Impact of data quality for automatic issue classification using pre-trained language models","venue":null,"work_id":"1ac4337a-d4c0-4368-96a8-4377c2882b89","year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.511431Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:2d803a54ffd84cac1008e51f7dcd2054e9e8bc80b18488d5b7aeb7bed9884f55","observation_id":"33546260-c2d9-4414-8686-4e4228eafcd3","resolution":{"observed_at":"2026-08-08T11:08:25.894416Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.880432Z","title":null,"venue":null,"work_id":"09aa324e-fc09-4c2c-88e0-81ae17b287be","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.515980Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:758394c713c709977673504a0c5107e44b3e6dcd7c65e217a7548c5449f5e5b4","observation_id":"746d299c-d1c1-4d6a-a4a3-c128dbb24904","resolution":{"observed_at":"2026-08-08T11:08:25.884467Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.809349Z","title":null,"venue":null,"work_id":"f8dd05e6-f69f-42b2-b985-ece078ffc76e","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.520306Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7d8be74bced11d09f17706af58011c447cb52813351d99718d59f426ceb3ad1e","observation_id":"44d1242c-0a2b-441a-8aaf-727861819fba","resolution":{"observed_at":"2026-08-08T11:08:25.859689Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.587782Z","title":null,"venue":null,"work_id":"031c9632-45fd-4d0d-ac11-c97c30a32abf","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.523529Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:48623775aceb36c28fec41ba9a9c2cf49dd8f3d8a6b79e0f51ee7a28ecbb679c","observation_id":"26ad1efa-fb80-479f-bfa8-0c9ef9f67e88","resolution":{"observed_at":"2026-08-08T11:08:25.705538Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.527609Z","title":"A Comprehensive Survey of Grammatical Error Correction","venue":null,"work_id":"41f6bf5a-bffa-43ec-84df-e3a864455063","year":2021},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.527345Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:adb94a0b8756245bee06af364e72855a71283390cdd0db991abeef577bc48273","observation_id":"292818a7-0ad8-43b1-a081-0088a475d3b8","resolution":{"observed_at":"2026-08-08T11:08:25.540618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.514414Z","title":"The CoNLL-2014 Shared Task on Grammatical Error Correction","venue":null,"work_id":"3644933f-adfd-4ae5-8eb5-8f2c2100e072","year":2014},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.530256Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:3e6c44afca87623ef21a147d7ab117b6a80eb836531bcc72c34e7d77ab6fcbb1","observation_id":"bf769fc2-7486-410f-a78e-6ba500cd2d87","resolution":{"observed_at":"2026-08-08T11:08:25.519429Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.501419Z","title":"Grammatical Error Correction: A Survey of the State of the Art","venue":null,"work_id":"bbbf446d-8daa-4cdf-8144-c5dc547ba8fd","year":2023},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.533729Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:f0b23fbaae4f8ea69dd4e3e4ecf21a4b7c830ce70b6a125ef517ab047d5749c4","observation_id":"6adabad0-8498-40c9-891d-ddb3f1aff8ee","resolution":{"observed_at":"2026-08-08T11:08:25.505767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18344","last_updated":"2024-05-28T16:42:43Z","snapshot_observed_at":"2026-07-06T18:21:26.195207Z","submitted_at":"2024-05-28T16:42:43Z","title":"The Battle of LLMs: A Comparative Study in Conversational QA Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.18344","snapshot_observed_at":"2026-08-08T11:08:24.538142Z","title":"The Battle of LLMs: A Comparative Study in Conversational QA Tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.538142Z"},"links":{"cited_paper":"/paper/2405.18344","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:907b44c7ffbc97e0283c49c3b001b09690c4b34665db3ae45e0badef44baf83d","observation_id":"7bc03968-b719-4430-ad1a-6ae77a1130c7","resolution":{"observed_at":"2026-08-08T11:08:24.538142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01001","last_updated":"2024-09-02T07:26:19Z","snapshot_observed_at":"2026-07-06T19:09:06.022484Z","submitted_at":"2024-09-02T07:26:19Z","title":"Beyond ChatGPT: Enhancing Software Quality Assurance Tasks with Diverse LLMs and Validation Techniques","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01001","snapshot_observed_at":"2026-08-08T11:08:24.623523Z","title":"Beyond ChatGPT: Enhancing Software Quality Assurance Tasks with Diverse LLMs and Validation Techniques","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.623523Z"},"links":{"cited_paper":"/paper/2409.01001","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7f3691542426a6020132aaf5deb73234812572628a9beb93df543ff723d28c15","observation_id":"16738db4-9576-4802-bade-c91faac82f45","resolution":{"observed_at":"2026-08-08T11:08:24.623523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-08T06:16:25.839566Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-08T11:08:24.722196Z","title":"Mixtral of Experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.722196Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:73557612ca2e6557442d88e4329b27b356e6212f072c8934a5cea0e3cbc85a45","observation_id":"e5c7d7a6-795d-49bd-84bf-93796f4afb13","resolution":{"observed_at":"2026-08-08T11:08:24.722196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.490302Z","title":"Unstructured clinical notes within the 24 hours since admission predict short, mid & long-term mortality in adult ICU patients","venue":null,"work_id":"effbc655-848b-4340-a184-5ec0a7de5751","year":2022},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.785985Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:d4841da5eee0f5aa8df4073854ad3472ba86a587cd6c2843c6addea70e22eeec","observation_id":"58f3cdef-5725-444c-b569-215da995a68a","resolution":{"observed_at":"2026-08-08T11:08:25.494327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.479748Z","title":null,"venue":null,"work_id":"cb2bb63a-a933-46e4-a35b-a69834acf4c4","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.821466Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:c36ff59a999b57d117b1d5aa3553501840c804f8eebf2cd41b8bac62138db9ca","observation_id":"bb4e021a-4630-45f6-a0d3-62f570be6037","resolution":{"observed_at":"2026-08-08T11:08:25.483346Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:24.825014Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.825014Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:eed788873caa9a363ffdd02ad3daaed8f602ac93f44857c8dbfa571b0a97935b","observation_id":"5a2f893b-5f9f-40b0-87ce-633f23f9f552","resolution":{"observed_at":"2026-08-08T11:08:24.825014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.470125Z","title":"Distributed representations of words and phrases and their compositionality","venue":null,"work_id":"f54fb0ab-8a6f-4359-a598-388d8f0d4355","year":2013},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.828231Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:9bfc4d07c224d358b1bb9073a26102c36f65d72b7bc8075d4bf2adabb6ce3f9c","observation_id":"94555944-28c1-4cca-849e-f06dd318712f","resolution":{"observed_at":"2026-08-08T11:08:25.473547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.460123Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","venue":null,"work_id":"b848a76e-7cad-45b4-9d93-5ea4e0daded1","year":2019},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.831017Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:0d63751a8f178abff613f9fa9c5e3a30cc99b2f10aa1f410d8d3fad8919c8e95","observation_id":"4a7383fb-ca6a-408a-822d-656ac752d9ee","resolution":{"observed_at":"2026-08-08T11:08:25.463566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.448436Z","title":"Publicly Available Clinical BERT Embeddings","venue":null,"work_id":"07f59c14-3bc7-4408-88c2-4089fb8ae87c","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.835063Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:6cadda476f07bdacf4cfa70e143c2eb87d5fe16850059a40613546155194b2ea","observation_id":"150edee6-c040-4b05-86bb-7315896ed514","resolution":{"observed_at":"2026-08-08T11:08:25.452659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-08T11:08:24.838946Z","title":"Longformer: The Long-Document Transformer","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.838946Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:7dc9e2a548de91d39da7f6ed20dcb9533cffdc9214b96fc26bc3c49b8effc45c","observation_id":"9553200d-332b-48c1-82b3-ec60d4b58133","resolution":{"observed_at":"2026-08-08T11:08:24.838946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.436739Z","title":"Resident buzzed at2300hrs on5/10/10,her legs felt like they were burning n she was in pain","venue":null,"work_id":"a2a5aa32-5f2c-45bf-8a70-80dd9b942158","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.843684Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:2b6b7e100fed3fc73c3e091ff6302ffed0ec4f47e6ebb9675ce29d01a1046912","observation_id":"120725a8-17ec-4b71-b19c-be2188833d25","resolution":{"observed_at":"2026-08-08T11:08:25.440910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.410035Z","title":"Resident was sleeping on round check,repositioned by2 x staff fluids given. nil problems settled ator","venue":null,"work_id":"c0f82e37-9ee1-49ec-9386-ee7a3661f8f9","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.847927Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:0387a55e00f2e4145c72ab425bad8b9a51c901866f5f43160a8a3d32697fd631","observation_id":"99a3e58b-a47f-41ed-8f91-f78c40971625","resolution":{"observed_at":"2026-08-08T11:08:25.429295Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.217028Z","title":"Resident","venue":null,"work_id":"3cbf7194-572d-412d-b8eb-e4014870b08b","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.851824Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:76a0099436d51a51aba401142bd8fed7f807d06c84e90a6988cb03ba1d058cf6","observation_id":"099afcfe-dfc1-4c95-8c6d-3e075e0da905","resolution":{"observed_at":"2026-08-08T11:08:25.271864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T11:08:25.202867Z","title":"Resident","venue":null,"work_id":"fda9e21e-b0c6-4b70-b895-d46cc61a5ff7","year":null},"citing_paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-08T11:08:24.855494Z"},"links":{"citing_paper":"/paper/2502.08669"},"observation_digest":"sha256:24aa20e21720bd6af6acd13c91f9c02e219890346f4eed70fb601fa649288fda","observation_id":"d51656ab-a47b-442d-b9d1-f368a6d055fc","resolution":{"observed_at":"2026-08-08T11:08:25.207666Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2502.08669","last_updated":"2025-02-12T00:27:49Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T17:36:21.919755Z","submitted_at":"2025-02-12T00:27:49Z","title":"Assessing the Impact of the Quality of Textual Data on Feature Representation and Machine Learning Models"},"reference_resolution":{"displayed":73,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":15,"verified_exact":0,"verified_fuzzy":58},"total_outbound_references":73},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 73 of 73 outbound references and 0 inbound Pith citation observations for arXiv:2502.08669."}