{"as_of":"2026-08-13T08:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2fe2324cd5aa1852d42c2dfa576180f55c92072ff2438113f58974715ad98116","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T13:19:19.123213Z","state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:04:55.940338Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-07T05:04:55.940338Z","title":"Power, A., Burda, Y ., Edwards, H., Babuschkin, I., and Misra, V","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09099","last_updated":"2025-06-17T19:17:29Z","snapshot_observed_at":"2026-08-09T23:42:25.878848Z","submitted_at":"2025-06-10T14:49:33Z","title":"Too Big to Think: Capacity, Memorization, and Generalization in Pre-Trained Transformers","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T05:04:55.940338Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2506.09099"},"observation_digest":"sha256:b352bccdafd889fc1d24345b82fe7e018c4bbeb820faa56be4d7119b33f93eaf","observation_id":"e040c504-c48e-47f9-8502-6c11da144c51","resolution":{"observed_at":"2026-08-07T05:04:55.940338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-06T14:58:43.493356Z","title":"Preprint, arXiv:2412.13337","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.17204","last_updated":"2025-07-23T04:52:58Z","snapshot_observed_at":"2026-08-11T08:04:18.265008Z","submitted_at":"2025-07-23T04:52:58Z","title":"Filter-And-Refine: A MLLM Based Cascade System for Industrial-Scale Video Content Moderation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T14:58:43.493356Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2507.17204"},"observation_digest":"sha256:c9e338ea81ec85f80513867d26dd4e98488af64921d9d997239a5cb1725cd042","observation_id":"afba8959-19d8-42b6-9574-00a5845b1a6e","resolution":{"observed_at":"2026-08-06T14:58:43.493356Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-05T16:30:44.745578Z","title":"arXiv preprint arXiv:2412.13337 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.18444","last_updated":"2026-05-26T14:37:13Z","snapshot_observed_at":"2026-08-05T16:30:44.040752Z","submitted_at":"2025-08-25T19:48:39Z","title":"How Reliable are LLMs for Reasoning on the Re-ranking task?","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-05T16:30:44.745578Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2508.18444"},"observation_digest":"sha256:d1bb40f9a7eb64e06c13fcf40d08f2225afda0db827ce720cbe07d944ab8848f","observation_id":"9a21d251-f405-4f7e-8753-18363244b8e2","resolution":{"observed_at":"2026-08-05T16:30:44.745578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":"2412.13337","doi":"10.48550/arxiv.2412.13337","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Unveiling the secret recipe: A guide for supervised ﬁne-tuning small llms","venue":"arXiv (Cornell University)","work_id":"a8be75b4-fde1-475d-a569-d25f0a73b76b","year":2024},"citing_paper":{"arxiv_id":"2509.07177","last_updated":"2026-04-14T15:07:25Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-08T19:48:52Z","title":"Towards EnergyGPT: A Large Language Model Specialized for the Energy Sector","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-18T17:39:17.456350Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2509.07177"},"observation_digest":"sha256:0530f9c10899bc8a40dd4640f98a1ceeb03d286772c5c2f05faaf51c9e3038d0","observation_id":"54fdf153-b9f3-44e4-87d4-e4de3aec2fc7","resolution":{"observed_at":"2026-05-18T17:42:47.526644Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":"2412.13337","doi":"10.48550/arxiv.2412.13337","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Unveiling the secret recipe: A guide for supervised ﬁne-tuning small llms","venue":"arXiv (Cornell University)","work_id":"a8be75b4-fde1-475d-a569-d25f0a73b76b","year":2024},"citing_paper":{"arxiv_id":"2509.13047","last_updated":"2026-04-12T16:39:21Z","snapshot_observed_at":"2026-07-06T22:30:02.223630Z","submitted_at":"2025-09-16T13:04:48Z","title":"Multi-Model Synthetic Training for Mission-Critical Small Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-18T15:47:00.455107Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2509.13047"},"observation_digest":"sha256:1b0d2988ccc2328bb3215fb0d52ba8d09b678187935f008b991e79bc2aed8dc2","observation_id":"018f5302-5807-4622-8b25-cc6325cd7ad8","resolution":{"observed_at":"2026-05-18T15:51:34.943976Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":"2412.13337","doi":"10.48550/arxiv.2412.13337","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Unveiling the secret recipe: A guide for supervised ﬁne-tuning small llms","venue":"arXiv (Cornell University)","work_id":"a8be75b4-fde1-475d-a569-d25f0a73b76b","year":2024},"citing_paper":{"arxiv_id":"2512.03053","last_updated":"2026-04-16T21:39:03Z","snapshot_observed_at":"2026-08-13T06:45:24.232532Z","submitted_at":"2025-11-25T00:47:02Z","title":"Mitigating hallucinations and omissions in LLMs for invertible problems: An application to hardware logic design automation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T05:24:05.830241Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2512.03053"},"observation_digest":"sha256:180ed9dbeb43355c4049373df2bb3ec861ebc8fa95a2a0f6fc875d2ce224368d","observation_id":"070dbe26-0e03-41ef-bf26-b1e3584d6609","resolution":{"observed_at":"2026-05-17T05:29:05.118255Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":"2412.13337","doi":"10.48550/arxiv.2412.13337","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Unveiling the secret recipe: A guide for supervised ﬁne-tuning small llms","venue":"arXiv (Cornell University)","work_id":"a8be75b4-fde1-475d-a569-d25f0a73b76b","year":2024},"citing_paper":{"arxiv_id":"2601.22264","last_updated":"2026-04-04T04:37:18Z","snapshot_observed_at":"2026-08-02T23:44:33.684751Z","submitted_at":"2026-01-29T19:34:34Z","title":"Predicting Intermittent Job Failure Categories for Diagnosis Using Few-Shot Fine-Tuned Language Models","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T09:17:09.823817Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2601.22264"},"observation_digest":"sha256:01366328ef51f2369f9045eb9e938cf2660081c0c6b1d84532430e5b732e9226","observation_id":"47f62b80-dc44-4e9e-8269-1f95681662b5","resolution":{"observed_at":"2026-05-16T09:17:39.904727Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":"2412.13337","doi":"10.48550/arxiv.2412.13337","metadata_source":"arxiv_reference","pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Unveiling the secret recipe: A guide for supervised ﬁne-tuning small llms","venue":"arXiv (Cornell University)","work_id":"a8be75b4-fde1-475d-a569-d25f0a73b76b","year":2024},"citing_paper":{"arxiv_id":"2604.09791","last_updated":"2026-04-10T18:13:09Z","snapshot_observed_at":"2026-07-06T22:58:34.504325Z","submitted_at":"2026-04-10T18:13:09Z","title":"Pioneer Agent: Continual Improvement of Small Language Models in Production","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-10T17:48:40.520740Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2604.09791"},"observation_digest":"sha256:ec763b6ddffb8c747d4c2e43b66b82617e0d3056fbe1bd56812698781248bcb3","observation_id":"46622c25-0f5f-47b0-ae66-5f10df2355c9","resolution":{"observed_at":"2026-05-11T06:05:57.600419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.13337","snapshot_observed_at":"2026-08-02T13:56:49.004087Z","title":"Unveiling the secret recipe: A guide for supervised fine-tuning small llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18259","last_updated":"2026-05-15T15:34:48Z","snapshot_observed_at":"2026-08-08T01:48:17.759173Z","submitted_at":"2026-05-15T15:34:48Z","title":"Probabilistic Concept-Aware Steering for Trustworthy LLM Inference","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-02T13:56:49.004087Z"},"links":{"cited_paper":"/paper/2412.13337","citing_paper":"/paper/2607.18259"},"observation_digest":"sha256:ac8f8011abe06df83f75f1967c8ad936e11962ab84d64a4c1632db71742ccb79","observation_id":"afe23145-10fd-4949-ad1e-43ae8781db53","resolution":{"observed_at":"2026-08-02T13:56:49.004087Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.13337/citation-record","integrity":"/paper/2412.13337/integrity","json":"/paper/2412.13337/citation-record.json","paper":"/paper/2412.13337"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.14316","last_updated":"2024-07-16T10:22:51Z","snapshot_observed_at":"2026-08-10T16:01:01.114216Z","submitted_at":"2023-09-25T17:37:20Z","title":"Physics of Language Models: Part 3.1, Knowledge Storage and Extraction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14316","snapshot_observed_at":"2026-08-11T13:19:18.628389Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.628389Z"},"links":{"cited_paper":"/paper/2309.14316","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:587fd1f8c0c585b48a57411c605c8be7625900e9dc68b11757eda972ed6167e5","observation_id":"ae43ba94-cbcf-4f5a-b99d-35343de8bff3","resolution":{"observed_at":"2026-08-11T13:19:18.628389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10702","last_updated":"2023-11-20T02:01:33Z","snapshot_observed_at":"2026-08-13T05:22:58.620371Z","submitted_at":"2023-11-17T18:45:45Z","title":"Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10702","snapshot_observed_at":"2026-08-11T13:19:18.800381Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.800381Z"},"links":{"cited_paper":"/paper/2311.10702","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:c0014d756a7ae2d00cc336e65da1d5505fc1f15d6ffc45bd9ab21e109279e3a0","observation_id":"05f5428e-e25f-478b-8d07-f77c5fb1e409","resolution":{"observed_at":"2026-08-11T13:19:18.800381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.14491","last_updated":"2024-11-28T06:51:20Z","snapshot_observed_at":"2026-08-12T23:38:12.206478Z","submitted_at":"2024-06-20T16:55:33Z","title":"Instruction Pre-Training: Language Models are Supervised Multitask Learners","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.14491","snapshot_observed_at":"2026-08-11T13:19:18.655811Z","title":"In- struction pre-training: Language models are supervised multitask learners","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.655811Z"},"links":{"cited_paper":"/paper/2406.14491","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:8c51a36ad71d3aef7e952915df7cbcb6ca064684058d0b95b3c17878b0b96a43","observation_id":"8ef2ab61-229e-4258-8c46-7d31cd2f9faa","resolution":{"observed_at":"2026-08-11T13:19:18.655811Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.05457","last_updated":"2018-03-14T18:04:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2018-03-14T18:04:21Z","title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.05457","snapshot_observed_at":"2026-08-11T13:19:18.678386Z","title":"Think you have solved question answering? try arc, the ai2 reasoning challenge","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.678386Z"},"links":{"cited_paper":"/paper/1803.05457","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:2f38c771c5547b474cc4a9ffda26b2c679f82affb487984fe839a578770e59fe","observation_id":"6b8bf7df-a9ac-494b-9f91-b8b818631cda","resolution":{"observed_at":"2026-08-11T13:19:18.678386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"3126.7687","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:19.389041Z","title":"Both performed similarly, with stacked training slightly outperforming phased training across all bench- marks","venue":null,"work_id":"3c88b55c-7228-4c38-bd7a-1c29c67b6bd0","year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.091976Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:5b4b5ce5af3fe84544b331bd402f1e81a9ece9ac626506eda6ac9a5b9b3cee47","observation_id":"1d4f4b93-6746-4243-8c36-1e14399b683e","resolution":{"observed_at":"2026-08-11T13:19:19.399533Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:20.891494Z","title":"Phase Description # Samples Phase 00 Instruction following warmup: simple, template-based instruction-response pairs to transition the base models to instruction-following behavior","venue":null,"work_id":"dbeded3b-7b15-468c-9c95-eb9d1bf3057a","year":2024},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.086099Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:702f5ee740aa5482f70b5067f39863beb68a0f71187879b949c3094fa5e67bea","observation_id":"8dbc7d0a-7f59-49b9-90cd-fe8cb6c59e3c","resolution":{"observed_at":"2026-08-11T13:19:20.897460Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:18.722362Z","title":"Apple intelligence foundation language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.722362Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:19a6cdcfadaa4ee7ad01ecd0a50183bf379d37fa039a7b40a72c03be6f55641b","observation_id":"c781a5c7-c335-4b2d-9047-273f30a2422c","resolution":{"observed_at":"2026-08-11T13:19:18.722362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.18392","last_updated":"2024-10-17T12:01:15Z","snapshot_observed_at":"2026-08-12T23:55:55.131975Z","submitted_at":"2024-05-28T17:33:54Z","title":"Scaling Laws and Compute-Optimal Training Beyond Fixed Training Durations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.18392","snapshot_observed_at":"2026-08-11T13:19:18.736521Z","title":"Scaling laws and compute-optimal training beyond fixed training durations","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.736521Z"},"links":{"cited_paper":"/paper/2405.18392","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:b2fd1d990b4feb336276bdbb939af1e6446ee64a6a94bb65935093e756e55a67","observation_id":"06527b53-fad6-4c37-9a85-92dbe8aa7470","resolution":{"observed_at":"2026-08-11T13:19:18.736521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.09509","last_updated":"2022-07-14T13:04:29Z","snapshot_observed_at":"2026-07-06T12:49:18.486644Z","submitted_at":"2022-03-17T17:57:56Z","title":"ToxiGen: A Large-Scale Machine-Generated Dataset for Adversarial and Implicit Hate Speech Detection","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.09509","snapshot_observed_at":"2026-08-11T13:19:18.755939Z","title":"Toxigen: A large-scale machine-generated dataset for adversarial and implicit hate speech detec- tion","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.755939Z"},"links":{"cited_paper":"/paper/2203.09509","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:d8a316916409074cfe4c3e5b26f2a13bc5bc967b05aaf7b687f4986ccace74a6","observation_id":"63779e69-b0c9-47c8-b600-d2f5807b51f4","resolution":{"observed_at":"2026-08-11T13:19:18.755939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-10T12:35:09.020030Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-11T13:19:18.766781Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.766781Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:a3256acf9301ac47f1c90d2b53a44e845eeaeb03a399d619cc00a232d47d70fb","observation_id":"2d0ed83c-217d-4157-9df7-b015adf27b00","resolution":{"observed_at":"2026-08-11T13:19:18.766781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06395","last_updated":"2024-06-03T08:54:38Z","snapshot_observed_at":"2026-08-12T13:10:06.472493Z","submitted_at":"2024-04-09T15:36:50Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06395","snapshot_observed_at":"2026-08-11T13:19:18.778809Z","title":"Minicpm: Unveiling the potential of small language models with scalable training strategies","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.778809Z"},"links":{"cited_paper":"/paper/2404.06395","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:fba73aead4a2780c14a87c11dc1399eda45177dd968e3d42a882f35a75f72051","observation_id":"99f2e7e2-39f0-4541-a76b-8fcffdc9ccaa","resolution":{"observed_at":"2026-08-11T13:19:18.778809Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.08763","last_updated":"2024-09-04T16:13:18Z","snapshot_observed_at":"2026-08-13T00:54:55.069129Z","submitted_at":"2024-03-13T17:58:57Z","title":"Simple and Scalable Strategies to Continually Pre-train Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.08763","snapshot_observed_at":"2026-08-11T13:19:18.786182Z","title":"Adam Ibrahim, Benjamin Th ´erien, Kshitij Gupta, Mats L Richter, Quentin Anthony, Timoth ´ee Lesort, Eugene Belilovsky, and Irina Rish","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.786182Z"},"links":{"cited_paper":"/paper/2403.08763","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:d55260410601892466eca9e514f9234b115b432b1940df2633ec5cd17fc1a748","observation_id":"8fcd72c7-5c97-4060-ab9b-9ccbb8a3c978","resolution":{"observed_at":"2026-08-11T13:19:18.786182Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:20.833056Z","title":"This diversity reduces gradient variance, promoting stable updates and helping the model retain pre-trained knowledge without significant forgetting","venue":null,"work_id":"8967cf59-bda5-4f33-9ddd-c0a5728c933a","year":2023},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.117810Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:45c93eb2f13a790cd850346ae91df2768671b17b0a1377470558be4aafcce098","observation_id":"67a753df-c9fb-4dd0-ba04-67d9a6d956cf","resolution":{"observed_at":"2026-08-11T13:19:20.838699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-11T13:19:18.812023Z","title":"Mistral 7b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.812023Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:8b11b6c9f31d3790b5ddb5fe305e20584e913f2c791f9a1bc34c34cb6f78b914","observation_id":"317e3c17-0957-46ff-8d70-7939479d683b","resolution":{"observed_at":"2026-08-11T13:19:18.812023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.04088","last_updated":"2024-01-08T18:47:34Z","snapshot_observed_at":"2026-08-08T06:16:25.839566Z","submitted_at":"2024-01-08T18:47:34Z","title":"Mixtral of Experts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.04088","snapshot_observed_at":"2026-08-11T13:19:18.818157Z","title":"Mixtral of experts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.818157Z"},"links":{"cited_paper":"/paper/2401.04088","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:e1eac3b1d770e33309e9d3ba83d3f2bf9034dcdc427e4147018aabb8b2bd472b","observation_id":"3722976b-acc4-4abd-9f67-baaaa2202ef9","resolution":{"observed_at":"2026-08-11T13:19:18.818157Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.02178","last_updated":"2019-12-04T18:58:26Z","snapshot_observed_at":"2026-07-06T08:42:06.688732Z","submitted_at":"2019-12-04T18:58:26Z","title":"Fantastic Generalization Measures and Where to Find Them","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.02178","snapshot_observed_at":"2026-08-11T13:19:18.827430Z","title":"Fantastic generalization measures and where to find them","venue":null,"work_id":null,"year":1912},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.827430Z"},"links":{"cited_paper":"/paper/1912.02178","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:ae5ba893fc44937f8f46bf6364e843cdfbbc52898dac946eef79a8d2bfe96fcf","observation_id":"038ac8a1-9955-4e14-8302-7d1c7185c3ca","resolution":{"observed_at":"2026-08-11T13:19:18.827430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.01678","last_updated":"2022-07-10T22:05:29Z","snapshot_observed_at":"2026-08-09T13:40:47.411664Z","submitted_at":"2021-04-04T19:48:16Z","title":"Understanding Continual Learning Settings with Data Distribution Drift Analysis","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.01678","snapshot_observed_at":"2026-08-11T13:19:18.849324Z","title":"Timoth´ee Lesort, Massimo Caccia, and Irina Rish","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.849324Z"},"links":{"cited_paper":"/paper/2104.01678","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:3dc5118c1a0be9051176690ac5fefcd46fa278a20986950e6fa1810c51961de2","observation_id":"cfa9c493-f10c-406e-9d17-60bc22225a4d","resolution":{"observed_at":"2026-08-11T13:19:18.849324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12032","last_updated":"2024-04-06T03:52:04Z","snapshot_observed_at":"2026-08-09T18:45:15.884087Z","submitted_at":"2023-08-23T09:45:29Z","title":"From Quantity to Quality: Boosting LLM Performance with Self-Guided Data Selection for Instruction Tuning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12032","snapshot_observed_at":"2026-08-11T13:19:18.870117Z","title":"From quantity to quality: Boosting llm performance with self- guided data selection for instruction tuning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.870117Z"},"links":{"cited_paper":"/paper/2308.12032","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:70da3e3c836ed7820be0bc50262d3d14ea630a9ef98101479996cb06da1160c0","observation_id":"a4b61e5c-e1fd-4943-b860-3eee95e3bf3a","resolution":{"observed_at":"2026-08-11T13:19:18.870117Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.07958","last_updated":"2022-05-08T02:43:02Z","snapshot_observed_at":"2026-08-03T16:26:48.747700Z","submitted_at":"2021-09-08T17:15:27Z","title":"TruthfulQA: Measuring How Models Mimic Human Falsehoods","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.07958","snapshot_observed_at":"2026-08-11T13:19:18.876927Z","title":"Stephanie Lin, Jacob Hilton, and Owain Evans","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.876927Z"},"links":{"cited_paper":"/paper/2109.07958","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:5101778d866018984aee09209d3cd96c52ac0105a9fa43d0f12cefe93f53f2ce","observation_id":"b40ea2f7-6988-4959-a58b-74721d0550b7","resolution":{"observed_at":"2026-08-11T13:19:18.876927Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15685","last_updated":"2024-04-16T02:46:58Z","snapshot_observed_at":"2026-08-13T04:54:16.874411Z","submitted_at":"2023-12-25T10:29:28Z","title":"What Makes Good Data for Alignment? A Comprehensive Study of Automatic Data Selection in Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15685","snapshot_observed_at":"2026-08-11T13:19:18.883703Z","title":"What makes good data for alignment? a comprehensive study of automatic data selection in instruction tun- ing","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.883703Z"},"links":{"cited_paper":"/paper/2312.15685","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:cba8ccf54fc823e0c789eb538d4b34023eb694475468d726634675462778c72d","observation_id":"dbbfff60-5d2d-4a92-8c2a-9a207ab12da4","resolution":{"observed_at":"2026-08-11T13:19:18.883703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.13688","last_updated":"2023-02-14T16:33:33Z","snapshot_observed_at":"2026-08-05T20:18:14.325590Z","submitted_at":"2023-01-31T15:03:44Z","title":"The Flan Collection: Designing Data and Methods for Effective Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.13688","snapshot_observed_at":"2026-08-11T13:19:18.891682Z","title":"Shayne Longpre, Le Hou, Tu Vu, Albert Webson, Hyung Won Chung, Yi Tay, Denny Zhou, Quoc V Le, Barret Zoph, Jason Wei, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.891682Z"},"links":{"cited_paper":"/paper/2301.13688","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:4c94a8e0b242ffc9ee00e28a9846d8237b6781cb5d563adaf0e2b12b2fd7f39e","observation_id":"b44295e4-e54d-44c9-9db7-9f4020f34b5a","resolution":{"observed_at":"2026-08-11T13:19:18.891682Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04324","last_updated":"2024-05-07T13:50:40Z","snapshot_observed_at":"2026-08-13T00:13:05.937137Z","submitted_at":"2024-05-07T13:50:40Z","title":"Granite Code Models: A Family of Open Foundation Models for Code Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04324","snapshot_observed_at":"2026-08-11T13:19:18.898978Z","title":"Gran- ite code models: A family of open foundation models for code intelligence","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.898978Z"},"links":{"cited_paper":"/paper/2405.04324","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:6a891a5ce302c939304c56663324dd6ca607130980fd72b53ae768c83bcb7f92","observation_id":"c6efb0fa-c7be-4a3a-badb-16a9bf8251d9","resolution":{"observed_at":"2026-08-11T13:19:18.898978Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.11045","last_updated":"2023-11-21T19:43:31Z","snapshot_observed_at":"2026-08-13T05:22:43.282546Z","submitted_at":"2023-11-18T11:44:52Z","title":"Orca 2: Teaching Small Language Models How to Reason","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.11045","snapshot_observed_at":"2026-08-11T13:19:18.918005Z","title":"Arindam Mitra, Luciano Del Corro, Guoqing Zheng, Shweti Mahajan, Dany Rouhana, Andres Co- das, Yadong Lu, Wei-ge Chen, Olga Vrousgos, Corby Rosset, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.918005Z"},"links":{"cited_paper":"/paper/2311.11045","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:81c5494d0e273587aff698d9cea12b98e890354475c7396071b173624680390d","observation_id":"daa68dfe-25d1-44c7-9646-12e318b4f662","resolution":{"observed_at":"2026-08-11T13:19:18.918005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02707","last_updated":"2023-06-05T08:58:39Z","snapshot_observed_at":"2026-08-09T03:29:22.849759Z","submitted_at":"2023-06-05T08:58:39Z","title":"Orca: Progressive Learning from Complex Explanation Traces of GPT-4","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02707","snapshot_observed_at":"2026-08-11T13:19:18.927083Z","title":"Orca: Progressive learning from complex explanation traces of gpt-4","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.927083Z"},"links":{"cited_paper":"/paper/2306.02707","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:b23150874af0e89ca6464c497e5b987ca41a3e7bce0c48d15e4616a5ea47282d","observation_id":"27027e1d-b450-4e33-96f9-165096c0ff5c","resolution":{"observed_at":"2026-08-11T13:19:18.927083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:18.933396Z","title":"ISBN 9781450384421","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.933396Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:6fe315632569173b587e3a18f4eb5bc6e771bc6eb1efb8f732e5053e45be8d3c","observation_id":"80d76ed2-a615-4e87-90d1-d2f5d47abfda","resolution":{"observed_at":"2026-08-11T13:19:18.933396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T13:19:18.938992Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.938992Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:ac17a755c2a12e65039d30ff9a99f565d87e610b2ede1e224d61fa30b8932430","observation_id":"201fb40f-38f7-4224-a236-ca10c4388bbf","resolution":{"observed_at":"2026-08-11T13:19:18.938992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.02155","last_updated":"2022-03-04T07:04:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-04T07:04:42Z","title":"Training language models to follow instructions with human feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.02155","snapshot_observed_at":"2026-08-11T13:19:18.944956Z","title":"Wainwright, Pamela Mishkin, Chong Zhang, Sandhini Agarwal, Katarina Slama, Alex Ray, John Schulman, Jacob Hilton, Fraser Kel- ton, Luke E","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.944956Z"},"links":{"cited_paper":"/paper/2203.02155","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:98e51c1f0abb80b459043d94de04d3644d8263fc1f7f9af583ae09b96a5fc6ed","observation_id":"afb8f0e5-4615-49b3-a511-c7b0d5ed7b48","resolution":{"observed_at":"2026-08-11T13:19:18.944956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:20.931114Z","title":"Wei Pang, Chuan Zhou, Xiao-Hua Zhou, and Xiaojie Wang","venue":null,"work_id":"594a8817-ddba-43ce-86b1-da60ddec5aa1","year":2024},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.950730Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:903ea2c1a04b7f6efde7b5d807e3a959ce90bd72d958b21862fb820214c34ab5","observation_id":"03e83027-c86e-425b-b72f-9d79232e2c1a","resolution":{"observed_at":"2026-08-11T13:19:20.936609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-11T13:19:18.958987Z","title":"doi: 10.18653/v1/2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.958987Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:49f223990a648a3572564151cc24d6486ffe36583ab831a8552361467a16360b","observation_id":"19d08b4e-a8f1-43af-85a8-5f19d58653fc","resolution":{"observed_at":"2026-08-11T13:19:18.958987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.12673","last_updated":"2019-12-20T18:20:34Z","snapshot_observed_at":"2026-08-12T08:28:18.894363Z","submitted_at":"2019-09-27T13:27:53Z","title":"A Constructive Prediction of the Generalization Error Across Scales","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.12673","snapshot_observed_at":"2026-08-11T13:19:18.972099Z","title":"Jonathan S","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.972099Z"},"links":{"cited_paper":"/paper/1909.12673","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:92dbf8f364428770c2238e52b8506c4ef43f91a1cc2d0ab85279c2a755c6e8a2","observation_id":"cfa491ba-3343-4c2a-bc7c-91df6afcd161","resolution":{"observed_at":"2026-08-11T13:19:18.972099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.08207","last_updated":"2022-03-17T17:53:01Z","snapshot_observed_at":"2026-07-06T11:58:21.596920Z","submitted_at":"2021-10-15T17:08:57Z","title":"Multitask Prompted Training Enables Zero-Shot Task Generalization","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.08207","snapshot_observed_at":"2026-08-11T13:19:18.979564Z","title":"semanticscholar.org/CorpusID:203592013","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.979564Z"},"links":{"cited_paper":"/paper/2110.08207","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:673e58e2802868dbd3afa8c40ec87911012526f1ea71405d38925a1e7911e702","observation_id":"a87de158-bc32-4130-b990-04fe78c7f5ec","resolution":{"observed_at":"2026-08-11T13:19:18.979564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2301.16790","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:19.791760Z","title":"Teven Le Scao et al","venue":null,"work_id":"220e998e-2a1a-43fa-9fc1-51182dadfc71","year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.986072Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:f4efacf4877aac8f81f90661087209f03516ef701cfd3e120e7e967aed071911","observation_id":"75126dd1-0758-4751-8059-1470925cacf0","resolution":{"observed_at":"2026-08-11T13:19:19.808324Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1711.00489","last_updated":"2018-02-24T00:16:12Z","snapshot_observed_at":"2026-08-10T10:50:51.995764Z","submitted_at":"2017-11-01T18:04:31Z","title":"Don't Decay the Learning Rate, Increase the Batch Size","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1711.00489","snapshot_observed_at":"2026-08-11T13:19:18.995228Z","title":"SL Smith","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.995228Z"},"links":{"cited_paper":"/paper/1711.00489","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:61e5a7b9bc491b8c8e924503d05cf90cbe9134f35e89a1a9cf621b6fcfd49274","observation_id":"0b818c71-9ec1-48f7-8fac-1af8eed94182","resolution":{"observed_at":"2026-08-11T13:19:18.995228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09261","last_updated":"2022-10-17T17:08:26Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-17T17:08:26Z","title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.09261","snapshot_observed_at":"2026-08-11T13:19:19.021871Z","title":"Challenging big-bench tasks and whether chain-of-thought can solve them","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.021871Z"},"links":{"cited_paper":"/paper/2210.09261","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:952c9548187fc43d7350c3596cce8b9be569df281e5459e8e1b7695eeaa4ce6a","observation_id":"d04ef4d7-7c82-4939-a46a-3e5571cc9b40","resolution":{"observed_at":"2026-08-11T13:19:19.021871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-11T13:19:19.028038Z","title":"Llama 2: Open founda- tion and fine-tuned chat models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.028038Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:736baa0dc27c2682437ad33a85ce1cf307bf29511a8f3ae769a743b5ce32813c","observation_id":"9350659d-d123-43c4-bdac-3334c86ab06d","resolution":{"observed_at":"2026-08-11T13:19:19.028038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08388","last_updated":"2023-03-15T06:24:01Z","snapshot_observed_at":"2026-07-06T15:03:30.911601Z","submitted_at":"2023-03-15T06:24:01Z","title":"A Chandra X-ray Survey of Optically Selected Close Galaxy Pairs: Unexpectedly Low Occupation of Active Galactic Nuclei","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08388","snapshot_observed_at":"2026-08-11T13:19:19.034377Z","title":"Code alpaca: Code instruction data from community contributions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.034377Z"},"links":{"cited_paper":"/paper/2303.08388","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:ff68e2bba4605e4ff94387a7fee830c1ba4d9d574e2d1c183997c21282b7e931","observation_id":"01bd5928-7c03-45c4-81a0-0f4805a68a06","resolution":{"observed_at":"2026-08-11T13:19:19.034377Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01574","last_updated":"2024-11-06T02:54:00Z","snapshot_observed_at":"2026-08-06T00:29:17.674418Z","submitted_at":"2024-06-03T17:53:00Z","title":"MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01574","snapshot_observed_at":"2026-08-11T13:19:19.040669Z","title":"How far can camels go? exploring the state of instruction tuning on open resources","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.040669Z"},"links":{"cited_paper":"/paper/2406.01574","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:abd391fa3e9a993751eb14f20cd519fd8e88768ce9ad643392afdd5619834371","observation_id":"aef8f542-85f1-4ec3-9a91-d3605d1e261f","resolution":{"observed_at":"2026-08-11T13:19:19.040669Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.01652","last_updated":"2022-02-08T20:26:45Z","snapshot_observed_at":"2026-08-13T01:55:08.127780Z","submitted_at":"2021-09-03T17:55:52Z","title":"Finetuned Language Models Are Zero-Shot Learners","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.01652","snapshot_observed_at":"2026-08-11T13:19:19.050734Z","title":"Finetuned language models are zero-shot learners.arXiv preprint arXiv:2109.01652,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.050734Z"},"links":{"cited_paper":"/paper/2109.01652","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:5db3646a939d5d3f194b134bcf3389752e6720f5374a4de868719391b528e2db","observation_id":"b93ee87d-df18-4135-aef4-030e7673ab26","resolution":{"observed_at":"2026-08-11T13:19:19.050734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.11903","last_updated":"2023-01-10T23:07:57Z","snapshot_observed_at":"2026-08-13T07:04:41.220509Z","submitted_at":"2022-01-28T02:33:07Z","title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.11903","snapshot_observed_at":"2026-08-11T13:19:19.055747Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.055747Z"},"links":{"cited_paper":"/paper/2201.11903","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:13079d18d6fc4c7bc73cda9df356c445c5e48b1ed473f299ced99a820d3e1cc7","observation_id":"5e27c432-41c6-4dfd-94b8-e96f3b295f4b","resolution":{"observed_at":"2026-08-11T13:19:19.055747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.12244","last_updated":"2025-05-27T06:49:09Z","snapshot_observed_at":"2026-08-13T02:06:18.586697Z","submitted_at":"2023-04-24T16:31:06Z","title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.12244","snapshot_observed_at":"2026-08-11T13:19:19.061453Z","title":"Can Xu, Qingfeng Sun, Kai Zheng, Xiubo Geng, Pu Zhao, Jiazhan Feng, Chongyang Tao, and Daxin Jiang","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.061453Z"},"links":{"cited_paper":"/paper/2304.12244","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:d2152d1f2ac996ff12d43dadf7ffbc5104e41939c9227a15d0a0225359737645","observation_id":"2ba1ff1e-d3da-4c8b-8b89-d1765f3876a3","resolution":{"observed_at":"2026-08-11T13:19:19.061453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.03466","last_updated":"2022-03-28T08:12:14Z","snapshot_observed_at":"2026-08-12T17:49:31.221284Z","submitted_at":"2022-03-07T15:37:35Z","title":"Tensor Programs V: Tuning Large Neural Networks via Zero-Shot Hyperparameter Transfer","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.03466","snapshot_observed_at":"2026-08-11T13:19:19.067368Z","title":"Tensor programs v: Tuning large neural networks via zero-shot hyperparameter transfer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.067368Z"},"links":{"cited_paper":"/paper/2203.03466","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:c436c306d66aa8c2476345a1490eecb1ef2d1d3596b021fb5eb4d0e3460ac7ac","observation_id":"bd19a69d-dd16-4171-abab-f17d37adbbe4","resolution":{"observed_at":"2026-08-11T13:19:19.067368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.07431","last_updated":"2024-10-03T13:07:25Z","snapshot_observed_at":"2026-08-12T22:46:50.099484Z","submitted_at":"2024-09-11T17:21:59Z","title":"Synthetic continued pretraining","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.07431","snapshot_observed_at":"2026-08-11T13:19:19.072603Z","title":"Synthetic continued pretraining","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.072603Z"},"links":{"cited_paper":"/paper/2409.07431","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:d5626f63ab36997172c1cc76ae947532f41a982e043f840c115c4eff2e8e773f","observation_id":"e30bbe58-080c-455b-a6c6-dfd4f0833a79","resolution":{"observed_at":"2026-08-11T13:19:19.072603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11206","last_updated":"2023-05-18T17:45:22Z","snapshot_observed_at":"2026-08-08T19:18:06.171048Z","submitted_at":"2023-05-18T17:45:22Z","title":"LIMA: Less Is More for Alignment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11206","snapshot_observed_at":"2026-08-11T13:19:19.078971Z","title":"Lima: Less is more for alignment","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.078971Z"},"links":{"cited_paper":"/paper/2305.11206","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:8784de56c1e30061878d69e18a34f3ebed93bfce24f2b5c772b1dd173659d440","observation_id":"07bb832d-8f96-4de3-ae93-151a5137b85e","resolution":{"observed_at":"2026-08-11T13:19:19.078971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:20.872784Z","title":null,"venue":null,"work_id":"993737d3-67a5-40a3-955b-2f67c2bbcc57","year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.098661Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:db8e83e26a644082fb4d17cf98e3e2a4df15d9fa5d3fd7410050cd503456daf3","observation_id":"80cfc113-1df7-42a7-88d2-5d0adbaf5bbf","resolution":{"observed_at":"2026-08-11T13:19:20.878471Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:20.852182Z","title":null,"venue":null,"work_id":"031eefa6-270a-4959-ac7a-bbf7bc528db7","year":2024},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.108410Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:f65eba84f980c389ecde3e23618661ec0ed6a6f8edbbeb5a3ed49f20e6363794","observation_id":"999315a6-9b8c-48f5-9839-63c9ecb2ea6d","resolution":{"observed_at":"2026-08-11T13:19:20.859315Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"5240.5280","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:19.289494Z","title":null,"venue":null,"work_id":"ff896b61-33f4-4a7b-9311-a1b1e5ced852","year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.123213Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:99259c2e15b2c9d586fc14b022c400634cff14dea43bd4b2cc5c7be54664f97a","observation_id":"f590b67d-127f-4ecc-b04e-309d43828e11","resolution":{"observed_at":"2026-08-11T13:19:19.299965Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:18.713947Z","title":"Suriya Gunasekar, Yi Zhang, Jyoti Aneja, Caio C´esar Teodoro Mendes, Allie Del Giorno, Sivakanth Gopi, Mojan Javaheripi, Piero Kauffmann, Gustavo de Rosa, Olli Saarikivi, et al","venue":null,"work_id":null,"year":1966},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":1966,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.713947Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:b4b2763cb60633a1465f7f691ba1614e9f6fc3611c7183ee5a3929f8b2f9e755","observation_id":"4bb46fd3-315b-4f1f-9fc4-16d1649059bd","resolution":{"observed_at":"2026-08-11T13:19:18.713947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.01081","last_updated":"2024-04-29T18:55:34Z","snapshot_observed_at":"2026-08-13T04:05:08.101953Z","submitted_at":"2024-03-02T03:48:37Z","title":"LAB: Large-Scale Alignment for ChatBots","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.01081","snapshot_observed_at":"2026-08-11T13:19:19.014062Z","title":"Lab: Large-scale alignment for chatbots","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2014,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.014062Z"},"links":{"cited_paper":"/paper/2403.01081","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:07dea9b9f6b7b94f005798ac2df3a3dfb2a86a03cd9f5b47049dbf52a4a419ee","observation_id":"27934eb6-1015-4910-926d-af31b7cb0035","resolution":{"observed_at":"2026-08-11T13:19:19.014062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:20.954947Z","title":"Scaling laws for downstream task performance of large language models","venue":null,"work_id":"e6e15c7e-e8f7-4c47-b4cc-bf28c6ae1f48","year":2024},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.793310Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:d8af31dbbc2d0f471a7351bd876ee21fd85e88196c72b7fff6eae04ddc0f9a96","observation_id":"1fcd025d-8d6a-4785-805a-ade83a0ad5f0","resolution":{"observed_at":"2026-08-11T13:19:20.961905Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1706.02677","last_updated":"2018-04-30T21:53:41Z","snapshot_observed_at":"2026-08-09T05:23:26.365677Z","submitted_at":"2017-06-08T16:51:53Z","title":"Accurate, Large Minibatch SGD: Training ImageNet in 1 Hour","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.02677","snapshot_observed_at":"2026-08-11T13:19:18.704482Z","title":"Accurate, large minibatch sgd: Training imagenet in 1 hour","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2016,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.704482Z"},"links":{"cited_paper":"/paper/1706.02677","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:4f1e7d5b6d3108c4034f4d74f27e463af98214c0fded8b7924f693d89b66dde1","observation_id":"fe0afd7b-b973-4067-9a08-721d6af60270","resolution":{"observed_at":"2026-08-11T13:19:18.704482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T13:19:19.004131Z","title":"Dropout: a simple way to prevent neural networks from overfitting","venue":null,"work_id":null,"year":1929},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:19.004131Z"},"links":{"citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:e74524c4e8f5ecad5ee01dbc73f37789fb5716fc9df2c77a68f798298d04afa9","observation_id":"f3053140-27e1-40e4-ac3f-c12eef180cbe","resolution":{"observed_at":"2026-08-11T13:19:19.004131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-11T13:19:18.688751Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.688751Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:79eede10f298a35301b0c79b18f9716e0c1762c4d71347dfd339174357605cd7","observation_id":"ec2043be-48d8-4b9b-93d6-a84968c80c65","resolution":{"observed_at":"2026-08-11T13:19:18.688751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-11T13:19:18.835652Z","title":"Brown, Benjamin Chess, Rewon Child, Scott Gray, Alec Radford, Jeff Wu, and Dario Amodei","venue":null,"work_id":null,"year":2001},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.835652Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:fe3cdb772df29f5c5f74ad87461f1d37d7a6344461a56b0ee872904254b0e9b9","observation_id":"8dea4143-f835-4c16-bb9b-163d09b842f2","resolution":{"observed_at":"2026-08-11T13:19:18.835652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1609.04836","last_updated":"2017-02-09T20:38:16Z","snapshot_observed_at":"2026-07-06T05:10:58.923264Z","submitted_at":"2016-09-15T20:03:06Z","title":"On Large-Batch Training for Deep Learning: Generalization Gap and Sharp Minima","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1609.04836","snapshot_observed_at":"2026-08-11T13:19:18.842637Z","title":"13 Nitish Shirish Keskar, Dheevatsa Mudigere, Jorge Nocedal, Mikhail Smelyanskiy, and Ping Tak Pe- ter Tang","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.842637Z"},"links":{"cited_paper":"/paper/1609.04836","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:7775009f919aba751e83a3f607bb3badf64aae0aa7d92718ba6f5893a06bcde3","observation_id":"1173cb8a-6bf0-4339-9cb8-eec92d58a186","resolution":{"observed_at":"2026-08-11T13:19:18.842637Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13064","last_updated":"2024-02-20T15:00:35Z","snapshot_observed_at":"2026-08-13T04:14:27.645630Z","submitted_at":"2024-02-20T15:00:35Z","title":"Synthetic Data (Almost) from Scratch: Generalized Instruction Tuning for Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13064","snapshot_observed_at":"2026-08-11T13:19:18.856098Z","title":"semanticscholar.org/CorpusID:233024916","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.856098Z"},"links":{"cited_paper":"/paper/2402.13064","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:88f6f6fe64cb25c45a13e3ad0012f5a82c9fbae4025b8a536ab7895e5beca942","observation_id":"97e7064a-9418-43f8-a26a-526b3c3c1011","resolution":{"observed_at":"2026-08-11T13:19:18.856098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.07327","last_updated":"2023-10-31T11:38:07Z","snapshot_observed_at":"2026-08-12T11:44:11.715086Z","submitted_at":"2023-04-14T18:01:29Z","title":"OpenAssistant Conversations -- Democratizing Large Language Model Alignment","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.07327","snapshot_observed_at":"2026-08-11T13:19:18.697794Z","title":"Enhancing chat language models by scaling high-quality instructional conversations","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.697794Z"},"links":{"cited_paper":"/paper/2304.07327","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:4b0430b662ae7e43db11f1eaf8841fa5aced0285ecfb76e858f40578c0d0553b","observation_id":"5b6f82bb-a009-4e8f-8d8a-a8403644cf80","resolution":{"observed_at":"2026-08-11T13:19:18.697794Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02954","last_updated":"2024-01-05T18:59:13Z","snapshot_observed_at":"2026-08-10T17:03:38.042994Z","submitted_at":"2024-01-05T18:59:13Z","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.02954","snapshot_observed_at":"2026-08-11T13:19:18.639273Z","title":"Deepseek llm: Scaling open-source language models with longtermism","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.639273Z"},"links":{"cited_paper":"/paper/2401.02954","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:1e20e0797bbff54f97679cccf8ec51ef3f8210e3e9d923da9a2533e2d304d992","observation_id":"7cdea624-089f-4643-a396-32502d825ab4","resolution":{"observed_at":"2026-08-11T13:19:18.639273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.11416","last_updated":"2022-12-06T21:39:48Z","snapshot_observed_at":"2026-07-06T14:08:18.855958Z","submitted_at":"2022-10-20T16:58:32Z","title":"Scaling Instruction-Finetuned Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.11416","snapshot_observed_at":"2026-08-11T13:19:18.668913Z","title":"Scaling instruction-finetuned language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-11T13:19:18.668913Z"},"links":{"cited_paper":"/paper/2210.11416","citing_paper":"/paper/2412.13337"},"observation_digest":"sha256:845baca0aa275737a529faa7a4f64cad384e8ae4b45e7c4cf1b3899074f39105","observation_id":"205c0304-1d3a-4eb4-9580-7c5725d57e0a","resolution":{"observed_at":"2026-08-11T13:19:18.668913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.13337","last_updated":"2024-12-17T21:16:59Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-11T13:11:55.344762Z","submitted_at":"2024-12-17T21:16:59Z","title":"Unveiling the Secret Recipe: A Guide For Supervised Fine-Tuning Small LLMs"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":51,"verified_exact":3,"verified_fuzzy":4},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 9 inbound Pith citation observations for arXiv:2412.13337."}