{"as_of":"2026-08-09T17:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a8e164eaa617bcf6a5c72c6fdee1d6eb9493d637da998b544b290d8803ba2876","coverage":[{"denominator":44,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":44,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-31T23:49:14.789992Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.23319/citation-record","integrity":"/paper/2607.23319/integrity","json":"/paper/2607.23319/citation-record.json","paper":"/paper/2607.23319"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:10.797345Z","title":"Neural machine translation of rare words with subword units","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:10.797345Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:01b09054bbd8c8a2e8112516384646765ff6f9a9cc04398390881feb69e9c0d6","observation_id":"d614ecf1-ec24-400d-abfc-def344aada7f","resolution":{"observed_at":"2026-07-31T23:49:10.797345Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:10.866403Z","title":"BERT: Pre-training of deep bidirectional transformers for language understanding","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:10.866403Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:ac3f4436d949d9faf09e4e4822dd1bbe86d9d1ace429807514fea5f13e8b1df6","observation_id":"6e403881-22e3-406c-871d-a9fb5121ec87","resolution":{"observed_at":"2026-07-31T23:49:10.866403Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:10.931354Z","title":"SentencePiece: A simple and language independent subword tokenizer and detokenizer for neural text processing","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:10.931354Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:05d040b4e77f176f31d1b8930dbb2a88c0f0720f86ee9b737f1cf432003a7016","observation_id":"17b5a600-ccdf-47ff-a91e-3b2c2b71c3d0","resolution":{"observed_at":"2026-07-31T23:49:10.931354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15425","last_updated":"2023-10-20T10:09:55Z","snapshot_observed_at":"2026-07-06T15:32:49.322000Z","submitted_at":"2023-05-17T14:17:57Z","title":"Language Model Tokenizers Introduce Unfairness Between Languages","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.15425","snapshot_observed_at":"2026-07-31T23:49:11.023254Z","title":"Torr, and Adel Bibi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.023254Z"},"links":{"cited_paper":"/paper/2305.15425","citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:ec46c2bc7235f0c67974d90de1fbc40f48da56bdd0564aea875d179783035c0f","observation_id":"8afba8f7-16bf-4d88-83a8-5fab1838c950","resolution":{"observed_at":"2026-07-31T23:49:11.023254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.107280Z","title":"A distributed platform for Sanskrit processing","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.107280Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:9b6c2a80fee3f1950d3c54d2ec184a5bc942ba74363c82aacc4816eb49189a70","observation_id":"882e771e-4b02-4828-9b3d-76a29ad1c9d4","resolution":{"observed_at":"2026-07-31T23:49:11.107280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.08144","last_updated":"2016-10-08T19:10:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2016-09-26T19:59:55Z","title":"Google's Neural Machine Translation System: Bridging the Gap between Human and Machine Translation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.08144","snapshot_observed_at":"2026-07-31T23:49:11.198955Z","title":"Le, Mohammad Norouzi, Wolfgang Macherey, Maxim Krikun, Yuan Cao, Qin Gao, Klaus Macherey, et al","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.198955Z"},"links":{"cited_paper":"/paper/1609.08144","citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:42da8ec46f3c3ebb502580ee9afe6d4b487e2de9faacd088989941ea7a1547bf","observation_id":"07b9e22a-4792-4129-9796-3c63cc1957ad","resolution":{"observed_at":"2026-07-31T23:49:11.198955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.246182Z","title":"Language models are unsupervised multitask learners.OpenAI blog, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.246182Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:49a67009af08d892bf2f91a34b7f3c1a8dc28d02e9afb4e34179b0899b286dd0","observation_id":"dd746852-4738-48ad-a10d-4a6c7553cf46","resolution":{"observed_at":"2026-07-31T23:49:11.246182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s40747-025-01780-5","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Research on morphological knowledge-guided low-resource agglutinative languages- chinese translation.Complex and Intelligent Systems, 11, 2025","venue":"Complex & Intelligent Systems","work_id":"eeb14507-e935-4c04-affa-e6b05dbe37ee","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.324785Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:4e8f1a808db2d7dbd56adca48b5f7e567c1dc3cb4ac0535f83cd02e705408017","observation_id":"995d4bf7-b8a7-4cb7-bc9e-8c09bfd4737a","resolution":{"observed_at":"2026-07-31T23:51:46.811799Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.30919/es2102","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Analysing unified embedding with morphological insight for multilingual text representation","venue":"Engineered Science","work_id":"8c31e8bf-d2ad-4f12-8958-7d508dcff86b","year":2026},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.407165Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:8560d9ce60fe9aaafae06dfbc3a4b5d9ea58b7ba5a87ff74ef28c452fad288df","observation_id":"0bfe206c-c069-4711-8806-52bd3f91175a","resolution":{"observed_at":"2026-07-31T23:51:46.740136Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.480850Z","title":"Unsupervised cross-lingual representation learning at scale","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.480850Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:71767978b1ad9e6b4f40460d117736d25a21c2b66ca708d487e2f95ae840a21b","observation_id":"91410b1e-c194-443e-8075-4adc58336991","resolution":{"observed_at":"2026-07-31T23:49:11.480850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.533199Z","title":"Khapra, and Pratyush Kumar","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.533199Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:d8235d6d04823e995263933eba1aebf337c7a6d7f9f32bc702aa81eed0a602ce","observation_id":"1f663dec-ce1b-4fc6-a3c2-cbb67bf76527","resolution":{"observed_at":"2026-07-31T23:49:11.533199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.647773Z","title":"Khapra, and Pratyush Kumar","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.647773Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:6434d3b9988b807e4ccaf93acb5fba1b8914a7a393442f48d10f12c89491de14","observation_id":"8ca9200a-ee92-4a4c-aef5-08dcbcae57af","resolution":{"observed_at":"2026-07-31T23:49:11.647773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.emnlp-main.1224","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tokenization and representation biases in multilingual models on dialectal nlp tasks","venue":null,"work_id":"34e16967-75d8-4b83-99fc-e496980b0c15","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.764690Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:9fd05c9ede58c4cc911187de2a51e17bb883acb891ca973bff584958d7dfe64b","observation_id":"11926be5-b701-46a0-b51a-7631b4e167b6","resolution":{"observed_at":"2026-07-31T23:51:46.628359Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.876057Z","title":"Kren-ne: A multilingual tokenization framework for northeast indian languages","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.876057Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:72a175e933b1da11f4090203423e5800326a9a11fbe8635057654b5771625118","observation_id":"148497e6-0a29-4f8a-8640-3a8122d1a904","resolution":{"observed_at":"2026-07-31T23:49:11.876057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:11.997430Z","title":"Domain-specific language model pretraining for biomedical natural language processing.ACM Transactions on Computing for Healthcare, 3(1):1–23, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:11.997430Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:23d2016cfed27b4f293a1e9256606cfdff9743b5914dafa3b586bd0048ec3a7b","observation_id":"89f847b9-0bad-4cb7-aabd-62373d520b85","resolution":{"observed_at":"2026-07-31T23:49:11.997430Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.078095Z","title":"LEGAL-BERT: The muppets straight out of law school","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.078095Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:fcdd83ba9615174f5964596c6f30962d80a572a9292dbc77ef352065e68cd0af","observation_id":"467f8ff6-b286-4a9d-84b7-6599968c9e51","resolution":{"observed_at":"2026-07-31T23:49:12.078095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.178977Z","title":"Pre- 30 training via paraphrasing","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.178977Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:e6abe1e071b540e536f25f4cb55f74f7f846f8f6f1929ad841ab1262a21dafef","observation_id":"c733f826-6029-4c37-a31b-826b3a7e142c","resolution":{"observed_at":"2026-07-31T23:49:12.178977Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.261171Z","title":"Word segmentation for classical Chinese: New standards and a study on using pre-training","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.261171Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:c9bb2e605f6e0ad1197c205659719cd39bf3a5e746eead20e896fd59097fcda5","observation_id":"f4a87f89-3c52-4edb-9962-cddedea3266f","resolution":{"observed_at":"2026-07-31T23:49:12.261171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.385847Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.385847Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:470266b2cb940d4775b294d209eb5925cbf9422bbcedb6a107f94d2189ef2f9c","observation_id":"c7c6ca47-68cc-49c7-8d2f-fb46e6bda83f","resolution":{"observed_at":"2026-07-31T23:49:12.385847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.476342Z","title":"Tokenization with factorized subword encoding.Findings of the Association for Computational Linguistics: ACL 2023, pages 14143–14161, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.476342Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:7c8a46e1de3c511cf560e1316010bd6e371a7d13267d5f5ff298e2c7f566f363","observation_id":"3fc05e64-4332-4041-8fad-a8faac0e4818","resolution":{"observed_at":"2026-07-31T23:49:12.476342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.565158Z","title":"Bpe gets picky: Efficient vocabulary refinement during tokenizer training.Proceedings of the 2024 Conference on Empirical Methods in Natural Language Processing, pages 16587–16604,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.565158Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:a413bbd67efc245fb7989df9abacb8384427c66b82e663a99ba74fd65dfd4e4b","observation_id":"9842564e-47b7-4332-aab7-01ad7a749efe","resolution":{"observed_at":"2026-07-31T23:49:12.565158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.762208Z","title":"Tokenization efficiency in code-switched text: Comparing sentencepiece and byte-pair encoding on taglish.TechRxiv Preprint, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.762208Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:3c57ac27850abb4657ae30d3a821b7e38267815a5bfce376c241f0ad9a95a198","observation_id":"96d877e0-47d8-4c89-afff-f8808c195ad2","resolution":{"observed_at":"2026-07-31T23:49:12.762208Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.860059Z","title":"Tokenization matters: Improving zero-shot ner for indic languages.IEEE International Conference on Electro Information Technology, pages 456–462, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.860059Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:a7ef7e5c1a8c2ac6c31b8d2fb54270e14e1ba9b8ec2d397606b833a859dcbdad","observation_id":"577f587a-72a3-4c3b-9ced-9faa31250f57","resolution":{"observed_at":"2026-07-31T23:49:12.860059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.969663Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.969663Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:540032e1882c10e69dbe8c8fe5741550578eb798bef581cc3be2165051a9b232","observation_id":"ea8bc7ba-2a47-446f-992c-5f20d802e72b","resolution":{"observed_at":"2026-07-31T23:49:12.969663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.acl-demo","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"4e09e54b-ab33-4cd5-a3a5-e0caede39cf8","year":2023},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.078696Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:2ad17647e90911cad7562ecd87590b8670e0eb31287a56f261729f4e8afc4a76","observation_id":"c834a339-d57d-48b6-a1ff-d944f6e9dbff","resolution":{"observed_at":"2026-07-31T23:51:46.432662Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3390/info17020128","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Morphology-aware segmentation and tokenization for turkic languages: A cse-guided framework (the kazakh case).Information Switzerland, 17, 2026","venue":"Information","work_id":"7f1a41ff-0fc3-4cb5-baa9-ab58ce06dfb6","year":2026},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.180110Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:57d7808410b83dbce2b0e6c6abb6762cc1922647b0b668db2441d62a16369840","observation_id":"2313deee-511a-43a2-bdf4-93b9870727ad","resolution":{"observed_at":"2026-07-31T23:51:46.341766Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2139/ssrn.5201887","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Multilingual tokenization efficiency in large language models: A study on indian languages.SSRN Electronic Journal, 2025","venue":"SSRN Electronic Journal","work_id":"dfd56066-6500-46c6-bbf4-c3d01b0bc19c","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.246179Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:2b2fc8f095d73297ab79b2cc98f9c910ad1d6a88f10891589f0bc971f25ed005","observation_id":"c6da959e-41d7-4944-a087-5873f327ba4a","resolution":{"observed_at":"2026-07-31T23:51:46.226773Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:13.352467Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.352467Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:efe286fe5af78968125865e6c30ca120e91940be8ab40f65c6c9f88114506239","observation_id":"b04c780d-883e-4566-8070-10a54a727fb6","resolution":{"observed_at":"2026-07-31T23:49:13.352467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/3705312","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","work_id":"f840b545-4f6a-441c-a067-fa50aed01f7e","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.459162Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:cd5b8cd9ab153dbaccd0be91f7d7a9b57c2eee7172e5d4a244205d94038b8e15","observation_id":"95360a38-5a25-421b-98cf-0d2507d01c4f","resolution":{"observed_at":"2026-07-31T23:51:46.118420Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:13.536123Z","title":"One model is all you need: Byt5-sanskrit, a unified model for sanskrit nlp tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.536123Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:b4eb0e22f835267998c66b9dabde2b6fa30327a932dc280b1c0d777825a19f75","observation_id":"7c015650-1a19-4e58-8375-5234a2a2217f","resolution":{"observed_at":"2026-07-31T23:49:13.536123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2026.eacl-srw.49","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"61397865-5ffe-4fbb-9e4f-61d0ee3c5606","year":2026},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.663301Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:9dda17e268b953b26732ce6e7315fd7ed90fca5ec0f1e92b8cbeabe650c714be","observation_id":"b98a1f47-173a-400e-96f6-da70f752f8f2","resolution":{"observed_at":"2026-07-31T23:51:46.028810Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.38124/ijisrt/25nov578","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"dc0ba0ac-68a9-464b-bbbb-7374cad30f9d","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.717042Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:8757b4333f4e3039517aed584cc04578a39e2b2c219a1bf317dfc4fc6a6f8189","observation_id":"ad5fb0b8-eb88-46f1-baa1-41a019d29962","resolution":{"observed_at":"2026-07-31T23:51:45.917306Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.acl-srw.57","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"bd9acd5f-758b-4dca-8f59-646d62028c3a","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.829541Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:86a0839c19e0ac3ff9a6363c390c9c3bea1f80afbba9e7533d313488e0b27c7a","observation_id":"020089dc-ce2f-4630-b943-2eac21d2151f","resolution":{"observed_at":"2026-07-31T23:51:45.830010Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:13.917084Z","title":"Multilingual denoising pre-training for neural machine translation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:13.917084Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:35a4caaa321bf1cc6bd6e4d3f7977bb4a0abfb83626c5a18b163feabda912fff","observation_id":"498204ec-20c2-4eff-8c95-4097e1311864","resolution":{"observed_at":"2026-07-31T23:49:13.917084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:14.019383Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.019383Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:218e85d54e16376fe81d5550c6152c3fbad98592960c568635796940e094df39","observation_id":"5f43fa6a-a45a-4cbf-ba5a-bbdda902b61d","resolution":{"observed_at":"2026-07-31T23:49:14.019383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.24818/ida-cl/2025.77","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nlp - tokenization and subword models.IDA-CL Working Papers, 2025","venue":null,"work_id":"82d13ad5-dbf9-46e7-aa81-c01168761502","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.217365Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:88a29ffc6f3a549a7e18aada86a8d99fda7aaa463bef99639be49f7de2833f7c","observation_id":"0601e604-c37d-438e-b087-45d3b70ed909","resolution":{"observed_at":"2026-07-31T23:51:45.747590Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s11042-024-20277-w","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":"Multimedia Tools and Applications","work_id":"15a65006-fdf2-4f43-a84a-76947c248286","year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.308553Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:f1b877d79153ae34e597eeeaed7e7884b6b022d4b9d241fbeb4f128239403ab3","observation_id":"79a8f0ae-ecd2-4bd3-a216-ae171744b4bf","resolution":{"observed_at":"2026-07-31T23:51:45.667817Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:14.418479Z","title":"A morphological tokenizer for generating vocabularies for large language models in spanish.Procesamiento del Lenguaje Natural, 75:29–40, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.418479Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:f8f464b93671c4d048179f6e8a7644ca499e5a3b4ce2318ff03c191f4b7c7b89","observation_id":"4e88418f-3e02-4a6f-bf15-567859509680","resolution":{"observed_at":"2026-07-31T23:49:14.418479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:14.529948Z","title":"Maibert: A pre-training corpus and language model for low-resourced maithili language","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.529948Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:211462ba931fa6384791c3b20ffd09ab4dfd5a7aa32bd1d6276e387031e90d33","observation_id":"c49a8047-1f32-4970-945e-13779433a3e9","resolution":{"observed_at":"2026-07-31T23:49:14.529948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3390/ai7020048","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spade-bert: Multilingual bert-based model with trigram-sensitive tokenization, tuned for depression detection in spanish texts.AI Switzerland, 7, 2026","venue":"AI","work_id":"ed540ea2-c5c9-42a9-adf1-d564e382c1b0","year":2026},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.594231Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:6c6a1e1b98dcfd55bf8649d51c0be8403f80a3b30264e5baec8c00a743146f86","observation_id":"705b1753-e636-4052-97dd-9394cf6faabd","resolution":{"observed_at":"2026-07-31T23:51:45.536664Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:14.676765Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.676765Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:068550ab1bddd23885dca31a3471187854adce00d440b20bc5e504fc4f1af17b","observation_id":"73ac79cb-9896-4754-ae53-d582cd9a2b16","resolution":{"observed_at":"2026-07-31T23:49:14.676765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:14.789992Z","title":"Analysis of subword tokenization approaches for turkish language.2023 31st Signal Processing and Communications Applications Conference (SIU), pages 1–4, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.789992Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:8d0a0afab4796c23e673b9713fa869c22de81bad3a671b5c23875f7d47468903","observation_id":"c939b22c-db04-4e6a-83ae-7ce7802008de","resolution":{"observed_at":"2026-07-31T23:49:14.789992Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:12.667922Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:12.667922Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:b32b179ab49b20f2c259640bda3970ba6226d825851142f7c233bf611f77180f","observation_id":"d4fcaf6f-bf8b-4232-b95c-5be7e22bad17","resolution":{"observed_at":"2026-07-31T23:49:12.667922Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-31T23:49:14.112646Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-07-31T23:49:14.112646Z"},"links":{"citing_paper":"/paper/2607.23319"},"observation_digest":"sha256:75202efcfbbeff1561b01a4a7e728b7a5290a381e18a854bfb76974ae16e4207","observation_id":"6e18440b-2d2e-4431-a852-fd6f7e504743","resolution":{"observed_at":"2026-07-31T23:49:14.112646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.23319","last_updated":"2026-07-25T18:23:06Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-31T23:49:10.129985Z","submitted_at":"2026-07-25T18:23:06Z","title":"BHARATI: Morphology-Aware Tokenizers for Classical Indian Languages with Subword Fertility Analysis"},"reference_resolution":{"displayed":44,"state_counts":{"malformed_identifier":4,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":27,"verified_exact":13,"verified_fuzzy":0},"total_outbound_references":44},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 44 of 44 outbound references and 0 inbound Pith citation observations for arXiv:2607.23319."}