{"as_of":"2026-08-13T03:03:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:dcce2f40f08378cc8be1423401fd820507a76e3dfb889d12fdadc5c7de7c3320","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T14:19:24.878747Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.15484/citation-record","integrity":"/paper/2411.15484/integrity","json":"/paper/2411.15484/citation-record.json","paper":"/paper/2411.15484"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.05829","last_updated":"2024-07-17T20:30:56Z","snapshot_observed_at":"2026-08-13T00:34:41.364835Z","submitted_at":"2024-04-08T19:48:36Z","title":"SambaLingo: Teaching Large Language Models New Languages","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05829","snapshot_observed_at":"2026-08-12T14:19:24.764296Z","title":"Preprint, arXiv:2404.05829","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.764296Z"},"links":{"cited_paper":"/paper/2404.05829","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:46cd1f6fd3315085b47b8cfacbfd1f162456e46f31eee88a80dfeab412c493e0","observation_id":"e00442f0-de60-4aa1-8571-364717310630","resolution":{"observed_at":"2026-08-12T14:19:24.764296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.267442Z","title":"In Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pages 3029–3051, Singapore","venue":null,"work_id":"4dbdf244-13c9-4635-a542-2e825de809b3","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.769934Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:79fca97e8cb69119bc462eec68b08016683302ec98bf5fb74fddff0c3a60f1ed","observation_id":"d6789c6c-f8f5-47cd-bb35-4a75d40d1959","resolution":{"observed_at":"2026-08-12T14:19:25.272550Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.15653","last_updated":"2023-11-27T09:33:13Z","snapshot_observed_at":"2026-08-11T01:24:02.284532Z","submitted_at":"2023-11-27T09:33:13Z","title":"MoDS: Model-oriented Data Selection for Instruction Tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.15653","snapshot_observed_at":"2026-08-12T14:19:24.774982Z","title":"Preprint, arXiv:2311.15653","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.774982Z"},"links":{"cited_paper":"/paper/2311.15653","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:4f79c90ece95772c514629dae251e86849094437ebc7d83f9ce082780570cb54","observation_id":"53879705-d94e-4c79-96f3-a4728a9e4047","resolution":{"observed_at":"2026-08-12T14:19:24.774982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13813","last_updated":"2024-04-22T01:22:23Z","snapshot_observed_at":"2026-08-13T00:25:05.118121Z","submitted_at":"2024-04-22T01:22:23Z","title":"From LLM to NMT: Advancing Low-Resource Machine Translation with Claude","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13813","snapshot_observed_at":"2026-08-12T14:19:24.780642Z","title":"Preprint, arXiv:2404.13813","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.780642Z"},"links":{"cited_paper":"/paper/2404.13813","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:9c7548840fd5dfe8341557d62e3ec10adb4733bf4e0d752c95eb4efa6354531d","observation_id":"5b27a6f9-8628-43ff-a85f-0a971db5a9d3","resolution":{"observed_at":"2026-08-12T14:19:24.780642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.249175Z","title":"In Findings of the As- sociation for Computational Linguistics: EMNLP 2023, pages 693–703, Singapore","venue":null,"work_id":"94bd3247-334a-4f3a-9a62-bfe3b9de0fa0","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.786262Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:cd6fdbcd12b4ab3c3717da42584e0fbd699db95f052cc880f581788930edfddf","observation_id":"a90beab7-f891-45e1-a036-7971cf5035ea","resolution":{"observed_at":"2026-08-12T14:19:25.255098Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.232092Z","title":"In Findings of the Association for Com- putational Linguistics: EMNLP 2023, pages 12365– 12394, Singapore","venue":null,"work_id":"96059641-2360-490d-a55e-7ccfd99671ba","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.791777Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:95d50e14ec72ac995bb8ab6acee7b59eb1b182865a62d5a332d132271537012a","observation_id":"5a573eef-1220-4d41-90b2-8c7b55ed3004","resolution":{"observed_at":"2026-08-12T14:19:25.237668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14517","last_updated":"2024-01-17T17:41:18Z","snapshot_observed_at":"2026-08-03T22:28:11.097742Z","submitted_at":"2023-09-25T20:23:51Z","title":"Watch Your Language: Investigating Content Moderation with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14517","snapshot_observed_at":"2026-08-12T14:19:24.796639Z","title":"Preprint, arXiv:2309.14517","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.796639Z"},"links":{"cited_paper":"/paper/2309.14517","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:9a4bedfd11ef54ca59202bb75b07cee473b2ad128e686f65578eb8b763bdf6ea","observation_id":"e41a716f-3739-4aaf-baa2-346e0ca0ce7e","resolution":{"observed_at":"2026-08-12T14:19:24.796639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13169","last_updated":"2023-11-13T14:50:06Z","snapshot_observed_at":"2026-08-12T15:01:41.972781Z","submitted_at":"2023-05-22T15:57:53Z","title":"A Pretrainer's Guide to Training Data: Measuring the Effects of Data Age, Domain Coverage, Quality, & Toxicity","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13169","snapshot_observed_at":"2026-08-12T14:19:24.802647Z","title":"Preprint, arXiv:2305.13169","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.802647Z"},"links":{"cited_paper":"/paper/2305.13169","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:9423fcefb766e865e9931e1638f002e5bf8cdac8f56143452cef1bbb96d35b76","observation_id":"826c3200-6c6b-496d-8e04-4463208f9ed1","resolution":{"observed_at":"2026-08-12T14:19:24.802647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12931","last_updated":"2024-04-30T21:35:53Z","snapshot_observed_at":"2026-08-02T10:40:03.816188Z","submitted_at":"2023-10-19T17:31:01Z","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12931","snapshot_observed_at":"2026-08-12T14:19:24.808542Z","title":"Preprint, arXiv:2310.12931","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.808542Z"},"links":{"cited_paper":"/paper/2310.12931","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:d999d621e1092ef92876561e2b537e7f243654136fbb5bfbc0878278ac2d7a38","observation_id":"52635de5-0b81-4be7-8a4b-6f4a6e7f814c","resolution":{"observed_at":"2026-08-12T14:19:24.808542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00738","last_updated":"2024-07-01T05:52:31Z","snapshot_observed_at":"2026-08-12T23:19:11.346529Z","submitted_at":"2023-12-01T17:17:56Z","title":"SeaLLMs -- Large Language Models for Southeast Asia","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00738","snapshot_observed_at":"2026-08-12T14:19:24.813951Z","title":"Preprint, arXiv:2312.00738","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.813951Z"},"links":{"cited_paper":"/paper/2312.00738","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:714ec48a5291740f22259a5ed826c99e3c4d57e4093d68a3dbd5cc8e67e9ac36","observation_id":"6c51e138-e74b-41a5-9afc-bbe9b391e0c4","resolution":{"observed_at":"2026-08-12T14:19:24.813951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-12T14:19:24.820133Z","title":"Preprint, arXiv:2303.08774","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.820133Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:0fd77c7ed8f9e06a6ac91e4f31a4d83ebaf799ec63287c57460d95f1768ac4c7","observation_id":"d6139071-8860-4c50-9e0e-fbbb55b1321b","resolution":{"observed_at":"2026-08-12T14:19:24.820133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16127","last_updated":"2024-04-23T12:31:30Z","snapshot_observed_at":"2026-08-13T00:47:00.507297Z","submitted_at":"2024-03-24T12:49:30Z","title":"WangchanLion and WangchanX MRC Eval","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16127","snapshot_observed_at":"2026-08-12T14:19:24.825966Z","title":"Preprint, arXiv:2403.16127","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.825966Z"},"links":{"cited_paper":"/paper/2403.16127","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:360e32d7186c647b36edd122fb33db1a8b4183853c5208456594317006e3b958","observation_id":"1595b964-ab38-4dcc-ad50-cf2456b136f9","resolution":{"observed_at":"2026-08-12T14:19:24.825966Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13951","last_updated":"2023-12-21T15:38:41Z","snapshot_observed_at":"2026-07-06T17:06:37.369542Z","submitted_at":"2023-12-21T15:38:41Z","title":"Typhoon: Thai Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.13951","snapshot_observed_at":"2026-08-12T14:19:24.831120Z","title":"Preprint, arXiv:2312.13951","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.831120Z"},"links":{"cited_paper":"/paper/2312.13951","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:d8900f608aa65457666b50b9028e3e9fd0e601a626fa566e8ca90a6078a41f98","observation_id":"e912cc9f-a8fc-46a3-8001-97eb5bcf5239","resolution":{"observed_at":"2026-08-12T14:19:24.831120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.215304Z","title":"In Proceedings of the First Workshop on Patient-Oriented Language Pro- cessing (CL4Health) @ LREC-COLING 2024, pages 124–130, Torino, Italia","venue":null,"work_id":"db454161-ac5d-4880-bd71-4ecc075b05b6","year":2024},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.836028Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:a004ccba483eaeb9ad180cbba4fbefcf0c365476b7c7af1e6b3df30b95b247b8","observation_id":"d91c6ed3-6955-468c-ac68-ff86ec8144b2","resolution":{"observed_at":"2026-08-12T14:19:25.220721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.198519Z","title":"In Findings of the Association for Computational Linguistics: EMNLP 2023, pages 1941–1961, Singapore","venue":null,"work_id":"25af12ae-94e5-4d1d-ba21-3aa71b73c58a","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.841248Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:abb0dea2240fd36a669018f39e753894a58e593c02b9d2c0db227c3d2c705fc1","observation_id":"a1b79abc-757d-422a-893e-04268e6cd960","resolution":{"observed_at":"2026-08-12T14:19:25.203944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-12T14:19:24.846881Z","title":"Preprint, arXiv:2312.11805","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.846881Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:7a588f84928a1ce1859424b1e6272e5094895119169eebba34e9b1677e9b06b9","observation_id":"f0ed2940-5a0c-417d-b877-f6234b84018c","resolution":{"observed_at":"2026-08-12T14:19:24.846881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-07-06T14:33:11.945106Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-08-12T14:19:24.857946Z","title":"Preprint, arXiv:2212.10560","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.857946Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:25184f9dc4e55d0032452fd258c9bcdad8a02939435fb2ae4610293b741d84fa","observation_id":"aa48736a-fcb9-4693-87da-68b5b7c5c5da","resolution":{"observed_at":"2026-08-12T14:19:24.857946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12244","last_updated":"2025-05-27T06:49:09Z","snapshot_observed_at":"2026-08-13T02:06:18.586697Z","submitted_at":"2023-04-24T16:31:06Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12244","snapshot_observed_at":"2026-08-12T14:19:24.863719Z","title":"Preprint, arXiv:2304.12244","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.863719Z"},"links":{"cited_paper":"/paper/2304.12244","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:2eddcd08f3359bceb073cd8a8fcadd3fc7f2b0a0931554220cef9abce98a8e36","observation_id":"49219387-37eb-4bb6-a27f-ba7b386abdb9","resolution":{"observed_at":"2026-08-12T14:19:24.863719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13606","last_updated":"2025-05-23T18:38:42Z","snapshot_observed_at":"2026-08-08T07:19:07.167593Z","submitted_at":"2024-02-21T08:20:06Z","title":"MlingConf: A Comprehensive Study of Multilingual Confidence Estimation on Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13606","snapshot_observed_at":"2026-08-12T14:19:24.869107Z","title":"Preprint, arXiv:2402.13606","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.869107Z"},"links":{"cited_paper":"/paper/2402.13606","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:e2f79d696d4784c81350519e26f653ef8b3e7a799a98be103c830021f2d9f0f6","observation_id":"24dcdebb-d9c8-495b-b2d7-51b996158bce","resolution":{"observed_at":"2026-08-12T14:19:24.869107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.178325Z","title":"In Proceedings of the 2023 Conference on Empirical Methods in Natu- ral Language Processing, pages 7915–7927, Singa- pore","venue":null,"work_id":"84592360-36ba-42be-9d1c-2f4ad5c11da1","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.873967Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:c123fe79a22c84362d63aeffc34e811445576c20e85685a00a39d64db82969e6","observation_id":"5ddcad77-e98d-4f4b-930a-8210cff89cfa","resolution":{"observed_at":"2026-08-12T14:19:25.186730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11206","last_updated":"2023-05-18T17:45:22Z","snapshot_observed_at":"2026-08-08T19:18:06.171048Z","submitted_at":"2023-05-18T17:45:22Z","title":"LIMA: Less Is More for Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11206","snapshot_observed_at":"2026-08-12T14:19:24.878747Z","title":"Preprint, arXiv:2305.11206","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.878747Z"},"links":{"cited_paper":"/paper/2305.11206","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:17fcece3e191ed41a258a3be80d85946767958ecd253da34e8b7a7fb73dd72e0","observation_id":"167b625b-0865-4a04-a8fb-e19620b65076","resolution":{"observed_at":"2026-08-12T14:19:24.878747Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.300294Z","title":null,"venue":null,"work_id":"da83c865-79eb-4bfb-a171-955459d56b5a","year":2019},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.747988Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:6b5ecb8f4d22788a55fbb5d0cc8155f704a5c204a56d9e26d08529954d37e976","observation_id":"979baa38-4f44-4c52-a8bd-ee6bc09e70ad","resolution":{"observed_at":"2026-08-12T14:19:25.305434Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2207.04672","last_updated":"2022-08-25T17:10:53Z","snapshot_observed_at":"2026-07-06T13:29:47.927628Z","submitted_at":"2022-07-11T07:33:36Z","title":"No Language Left Behind: Scaling Human-Centered Machine Translation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2207.04672","snapshot_observed_at":"2026-08-12T14:19:24.852629Z","title":"Preprint, arXiv:2207.04672","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.852629Z"},"links":{"cited_paper":"/paper/2207.04672","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:f7b0ef363e1ccd04d78bd641fdfda596b2723c98c9ecf4f41e284ef926080813","observation_id":"6d9232fb-e918-4b98-9a45-a5783fe511a8","resolution":{"observed_at":"2026-08-12T14:19:24.852629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:19:25.284806Z","title":"In Proceedings of the 2023 Conference on Empir- ical Methods in Natural Language Processing, pages 4232–4267, Singapore","venue":null,"work_id":"7c093f9d-2c7d-47e2-8f7f-817f2399f2d5","year":2023},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.753574Z"},"links":{"citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:66d576508faa52c44405ab54a23e22b435f0e63a0fe821c300ab40ba49ef035f","observation_id":"c5fbd0f1-23a5-42c1-9331-b8fa8d606f64","resolution":{"observed_at":"2026-08-12T14:19:25.290224Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03216","last_updated":"2025-12-12T11:26:32Z","snapshot_observed_at":"2026-07-06T17:25:35.493362Z","submitted_at":"2024-02-05T17:26:49Z","title":"M3-Embedding: Multi-Linguality, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03216","snapshot_observed_at":"2026-08-12T14:19:24.758805Z","title":"Preprint, arXiv:2402.03216","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T14:19:24.758805Z"},"links":{"cited_paper":"/paper/2402.03216","citing_paper":"/paper/2411.15484"},"observation_digest":"sha256:7ad44efcbcfe73285779ed6625b188e6eb36c0914742015d10514b42c9e52198","observation_id":"c3892b82-6df8-4a13-938e-10c929315e32","resolution":{"observed_at":"2026-08-12T14:19:24.758805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2411.15484","last_updated":"2024-11-23T07:50:59Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-12T14:12:36.942288Z","submitted_at":"2024-11-23T07:50:59Z","title":"Seed-Free Synthetic Data Generation Framework for Instruction-Tuning LLMs: A Case Study in Thai"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":0,"verified_fuzzy":7},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2411.15484."}