{"as_of":"2026-08-23T03:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6fd00719a8dae016c4daed3e4e37d47d5743eb19fdc03471f50b1a404776752a","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T14:19:24.878747Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.15484/citation-record","integrity":"/paper/2411.15484/integrity","json":"/paper/2411.15484/citation-record.json","paper":"/paper/2411.15484"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.05829","last_updated":"2024-07-17T20:30:56Z","snapshot_observed_at":"2026-08-16T14:02:36.334001Z","submitted_at":"2024-04-08T19:48:36Z","title":"SambaLingo: Teaching Large Language Models New Languages","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05829","snapshot_observed_at":"2026-08-12T14:19:24.764296Z","title":"Preprint, arXiv:2404.05829","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.764296Z"},"links":{"cited_paper":"/paper/2404.05829","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:bdf9230d3929d28f0bee342c7415653b27e3b62544a0b8e4301139773df62af2","observation_id":"e00442f0-de60-4aa1-8571-364717310630","resolution":{"observed_at":"2026-08-12T14:19:24.764296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.267442Z","title":"In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pages 3029–3051, Singapore","venue":null,"work_id":"4dbdf244-13c9-4635-a542-2e825de809b3","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.769934Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:99aa99983792571f386739c098ba22c3ab6d9c3c39baf89f782d0cb92d871277","observation_id":"d6789c6c-f8f5-47cd-bb35-4a75d40d1959","resolution":{"observed_at":"2026-08-12T14:19:25.272550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15653","last_updated":"2023-11-27T09:33:13Z","snapshot_observed_at":"2026-08-21T00:25:10.936086Z","submitted_at":"2023-11-27T09:33:13Z","title":"MoDS: Model-oriented Data Selection for Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15653","snapshot_observed_at":"2026-08-12T14:19:24.774982Z","title":"Preprint, arXiv:2311.15653","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.774982Z"},"links":{"cited_paper":"/paper/2311.15653","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:f0b8e3561c0d18c8929165b69fb57f133e71df7bf149522dc71a55bbc79c23bd","observation_id":"53879705-d94e-4c79-96f3-a4728a9e4047","resolution":{"observed_at":"2026-08-12T14:19:24.774982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13813","last_updated":"2024-04-22T01:22:23Z","snapshot_observed_at":"2026-08-16T13:59:03.810116Z","submitted_at":"2024-04-22T01:22:23Z","title":"From LLM to NMT: Advancing Low-Resource Machine Translation with Claude","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13813","snapshot_observed_at":"2026-08-12T14:19:24.780642Z","title":"Preprint, arXiv:2404.13813","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.780642Z"},"links":{"cited_paper":"/paper/2404.13813","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:adae0aa12e3723ebb388a13cb57d25343c7db7f13eae5871b1b1a0ddc29162d8","observation_id":"5b27a6f9-8628-43ff-a85f-0a971db5a9d3","resolution":{"observed_at":"2026-08-12T14:19:24.780642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.249175Z","title":"In Findings of the As- sociation for Computational Linguistics: EMNLP 2023, pages 693–703, Singapore","venue":null,"work_id":"94bd3247-334a-4f3a-9a62-bfe3b9de0fa0","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.786262Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:8bb5cb201728b99e3c18fe5e499dfc2ab1ad825fe92670c2c8239a6ea54ca22c","observation_id":"a90beab7-f891-45e1-a036-7971cf5035ea","resolution":{"observed_at":"2026-08-12T14:19:25.255098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.232092Z","title":"In Findings of the Association for Com- putational Linguistics: EMNLP 2023, pages 12365– 12394, Singapore","venue":null,"work_id":"96059641-2360-490d-a55e-7ccfd99671ba","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.791777Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:af9055d13efd5f2ab74fc9cc8ab23efe4439c76017cb58cd69868e6f6449c114","observation_id":"5a573eef-1220-4d41-90b2-8c7b55ed3004","resolution":{"observed_at":"2026-08-12T14:19:25.237668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14517","last_updated":"2024-01-17T17:41:18Z","snapshot_observed_at":"2026-08-20T07:41:49.235810Z","submitted_at":"2023-09-25T20:23:51Z","title":"Watch Your Language: Investigating Content Moderation with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14517","snapshot_observed_at":"2026-08-12T14:19:24.796639Z","title":"Preprint, arXiv:2309.14517","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.796639Z"},"links":{"cited_paper":"/paper/2309.14517","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:5a894bc7fab9723141607078d8c82f5fe31fe82d9978f40fb57b98a225be3474","observation_id":"e41a716f-3739-4aaf-baa2-346e0ca0ce7e","resolution":{"observed_at":"2026-08-12T14:19:24.796639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13169","last_updated":"2023-11-13T14:50:06Z","snapshot_observed_at":"2026-08-20T03:20:03.635456Z","submitted_at":"2023-05-22T15:57:53Z","title":"A Pretrainer's Guide to Training Data: Measuring the Effects of Data Age, Domain Coverage, Quality, & Toxicity","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13169","snapshot_observed_at":"2026-08-12T14:19:24.802647Z","title":"Preprint, arXiv:2305.13169","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.802647Z"},"links":{"cited_paper":"/paper/2305.13169","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:6b4a5335a80a177f3e2b3e918f4f630f97e4cfd3a5d8af04cd5e6f4122ee773b","observation_id":"826c3200-6c6b-496d-8e04-4463208f9ed1","resolution":{"observed_at":"2026-08-12T14:19:24.802647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12931","last_updated":"2024-04-30T21:35:53Z","snapshot_observed_at":"2026-08-02T10:40:03.816188Z","submitted_at":"2023-10-19T17:31:01Z","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12931","snapshot_observed_at":"2026-08-12T14:19:24.808542Z","title":"Preprint, arXiv:2310.12931","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.808542Z"},"links":{"cited_paper":"/paper/2310.12931","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:d999d621e1092ef92876561e2b537e7f243654136fbb5bfbc0878278ac2d7a38","observation_id":"52635de5-0b81-4be7-8a4b-6f4a6e7f814c","resolution":{"observed_at":"2026-08-12T14:19:24.808542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00738","last_updated":"2024-07-01T05:52:31Z","snapshot_observed_at":"2026-08-21T15:56:08.074773Z","submitted_at":"2023-12-01T17:17:56Z","title":"SeaLLMs -- Large Language Models for Southeast Asia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00738","snapshot_observed_at":"2026-08-12T14:19:24.813951Z","title":"Preprint, arXiv:2312.00738","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.813951Z"},"links":{"cited_paper":"/paper/2312.00738","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:22a73a248000fbca022320066c7768ec892183703d1d7186817b5daaa40a7c25","observation_id":"6c51e138-e74b-41a5-9afc-bbe9b391e0c4","resolution":{"observed_at":"2026-08-12T14:19:24.813951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T14:19:24.820133Z","title":"Preprint, arXiv:2303.08774","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.820133Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:16744cee80af370a4fe441bf507cda7d12dc3f254ea91ac8f653df068bb8ceed","observation_id":"d6139071-8860-4c50-9e0e-fbbb55b1321b","resolution":{"observed_at":"2026-08-12T14:19:24.820133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16127","last_updated":"2024-04-23T12:31:30Z","snapshot_observed_at":"2026-08-19T20:05:46.692908Z","submitted_at":"2024-03-24T12:49:30Z","title":"WangchanLion and WangchanX MRC Eval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16127","snapshot_observed_at":"2026-08-12T14:19:24.825966Z","title":"Preprint, arXiv:2403.16127","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.825966Z"},"links":{"cited_paper":"/paper/2403.16127","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:d5bf49c8aa3f95ed84e4b41c63c992174b11d6dbf44915c5ab390a5695c8bae7","observation_id":"1595b964-ab38-4dcc-ad50-cf2456b136f9","resolution":{"observed_at":"2026-08-12T14:19:24.825966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13951","last_updated":"2023-12-21T15:38:41Z","snapshot_observed_at":"2026-08-18T07:37:57.824307Z","submitted_at":"2023-12-21T15:38:41Z","title":"Typhoon: Thai Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13951","snapshot_observed_at":"2026-08-12T14:19:24.831120Z","title":"Preprint, arXiv:2312.13951","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.831120Z"},"links":{"cited_paper":"/paper/2312.13951","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:677ba544fb9bccdb3e2979995676c33b2324726d3871d3a9a6d0995e4080cd59","observation_id":"e912cc9f-a8fc-46a3-8001-97eb5bcf5239","resolution":{"observed_at":"2026-08-12T14:19:24.831120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.215304Z","title":"In Proceedings of the First Workshop on Patient-Oriented Language Pro- cessing (CL4Health) @ LREC-COLING 2024, pages 124–130, Torino, Italia","venue":null,"work_id":"db454161-ac5d-4880-bd71-4ecc075b05b6","year":2024},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.836028Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:5028982ee79884c8a27b0657668afc1c97a2a2ce4183ee5a698ed199cca9df14","observation_id":"d91c6ed3-6955-468c-ac68-ff86ec8144b2","resolution":{"observed_at":"2026-08-12T14:19:25.220721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.198519Z","title":"In Findings of the Association for Computational Linguistics: EMNLP 2023, pages 1941–1961, Singapore","venue":null,"work_id":"25af12ae-94e5-4d1d-ba21-3aa71b73c58a","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.841248Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:70cece10763d2a9e3ade643ff28498e866b0e55e54b40fb49b4e7939ca81758d","observation_id":"a1b79abc-757d-422a-893e-04268e6cd960","resolution":{"observed_at":"2026-08-12T14:19:25.203944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-08-20T18:27:04.837880Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-12T14:19:24.846881Z","title":"Preprint, arXiv:2312.11805","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.846881Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:c0385cdf861fdc0f0fb838da7c572c4c75515c2ded6f399a3f4d6fd07e4c858b","observation_id":"f0ed2940-5a0c-417d-b877-f6234b84018c","resolution":{"observed_at":"2026-08-12T14:19:24.846881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-08-19T16:38:51.329886Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-08-12T14:19:24.857946Z","title":"Preprint, arXiv:2212.10560","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.857946Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:4abe4531d5b103ca82507ffee74994b0bf5bd90ff6037916d10e43822c289b4c","observation_id":"aa48736a-fcb9-4693-87da-68b5b7c5c5da","resolution":{"observed_at":"2026-08-12T14:19:24.857946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12244","last_updated":"2025-05-27T06:49:09Z","snapshot_observed_at":"2026-08-13T02:06:18.586697Z","submitted_at":"2023-04-24T16:31:06Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12244","snapshot_observed_at":"2026-08-12T14:19:24.863719Z","title":"Preprint, arXiv:2304.12244","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.863719Z"},"links":{"cited_paper":"/paper/2304.12244","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:2eddcd08f3359bceb073cd8a8fcadd3fc7f2b0a0931554220cef9abce98a8e36","observation_id":"49219387-37eb-4bb6-a27f-ba7b386abdb9","resolution":{"observed_at":"2026-08-12T14:19:24.863719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13606","last_updated":"2025-05-23T18:38:42Z","snapshot_observed_at":"2026-08-16T14:16:41.577745Z","submitted_at":"2024-02-21T08:20:06Z","title":"MlingConf: A Comprehensive Study of Multilingual Confidence Estimation on Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13606","snapshot_observed_at":"2026-08-12T14:19:24.869107Z","title":"Preprint, arXiv:2402.13606","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.869107Z"},"links":{"cited_paper":"/paper/2402.13606","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:2ccbfcc2a7a8348aa47baf1836f3791a3d8b21b1d9a33f0d59af5717958f0d43","observation_id":"24dcdebb-d9c8-495b-b2d7-51b996158bce","resolution":{"observed_at":"2026-08-12T14:19:24.869107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.178325Z","title":"In Proceedings of the 2023 Conference on Empirical Methods in Natu- ral Language Processing, pages 7915–7927, Singa- pore","venue":null,"work_id":"84592360-36ba-42be-9d1c-2f4ad5c11da1","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.873967Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:ce931407c75b0152477f983bf5e030b9f62bb3fdf83248713a2a6e6683776e5e","observation_id":"5ddcad77-e98d-4f4b-930a-8210cff89cfa","resolution":{"observed_at":"2026-08-12T14:19:25.186730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11206","last_updated":"2023-05-18T17:45:22Z","snapshot_observed_at":"2026-08-08T19:18:06.171048Z","submitted_at":"2023-05-18T17:45:22Z","title":"LIMA: Less Is More for Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11206","snapshot_observed_at":"2026-08-12T14:19:24.878747Z","title":"Preprint, arXiv:2305.11206","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.878747Z"},"links":{"cited_paper":"/paper/2305.11206","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:17fcece3e191ed41a258a3be80d85946767958ecd253da34e8b7a7fb73dd72e0","observation_id":"167b625b-0865-4a04-a8fb-e19620b65076","resolution":{"observed_at":"2026-08-12T14:19:24.878747Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.300294Z","title":null,"venue":null,"work_id":"da83c865-79eb-4bfb-a171-955459d56b5a","year":2019},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.747988Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:cca7830d64e7b0fca22bba67089f3b54dfbc05b1c802e5927919392b63ee51cc","observation_id":"979baa38-4f44-4c52-a8bd-ee6bc09e70ad","resolution":{"observed_at":"2026-08-12T14:19:25.305434Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.04672","last_updated":"2022-08-25T17:10:53Z","snapshot_observed_at":"2026-07-06T13:29:47.927628Z","submitted_at":"2022-07-11T07:33:36Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.04672","snapshot_observed_at":"2026-08-12T14:19:24.852629Z","title":"Preprint, arXiv:2207.04672","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.852629Z"},"links":{"cited_paper":"/paper/2207.04672","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:f7b0ef363e1ccd04d78bd641fdfda596b2723c98c9ecf4f41e284ef926080813","observation_id":"6d9232fb-e918-4b98-9a45-a5783fe511a8","resolution":{"observed_at":"2026-08-12T14:19:24.852629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.284806Z","title":"In Proceedings of the 2023 Conference on Empir- ical Methods in Natural Language Processing, pages 4232–4267, Singapore","venue":null,"work_id":"7c093f9d-2c7d-47e2-8f7f-817f2399f2d5","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.753574Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:aa26f7d3a98e50345291f06dab07729268d6f9985b1a5bfdf28ecec8cde94a64","observation_id":"c5fbd0f1-23a5-42c1-9331-b8fa8d606f64","resolution":{"observed_at":"2026-08-12T14:19:25.290224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03216","last_updated":"2025-12-12T11:26:32Z","snapshot_observed_at":"2026-08-17T15:25:09.479397Z","submitted_at":"2024-02-05T17:26:49Z","title":"M3-Embedding: Multi-Linguality, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03216","snapshot_observed_at":"2026-08-12T14:19:24.758805Z","title":"Preprint, arXiv:2402.03216","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.758805Z"},"links":{"cited_paper":"/paper/2402.03216","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:2c1e086f8342feee5295a7bba4e1ada70eb37e22f2b5c2a12803340c9d234cbc","observation_id":"c3892b82-6df8-4a13-938e-10c929315e32","resolution":{"observed_at":"2026-08-12T14:19:24.758805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2411.15484."}