{"as_of":"2026-08-08T03:38:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f905a4386cc916b16f7e2d9baad5d1ac013f959a4dd8e205f3c147ee073f1dab","coverage":[{"denominator":64,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":64,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T12:03:42.484850Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.00741/citation-record","integrity":"/paper/2506.00741/integrity","json":"/paper/2506.00741/citation-record.json","paper":"/paper/2506.00741"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:34.567215Z","title":"Kgquiz: Evaluating the generalization of encoded knowledge in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:34.567215Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:92d35e9956934a221c50eab0194d641a281101a8b1de193f934450332acc6c9d","observation_id":"d016e96c-6595-4339-b0cc-cb96a723844b","resolution":{"observed_at":"2026-08-07T12:03:34.567215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07008","last_updated":"2026-06-01T01:28:08Z","snapshot_observed_at":"2026-08-06T02:41:44.773909Z","submitted_at":"2024-03-09T02:47:11Z","title":"AutoEval Done Right: Using Synthetic Data for Model Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07008","snapshot_observed_at":"2026-08-07T12:03:34.669899Z","title":"Au- toeval done right: Using synthetic data for model evaluation.arXiv preprint arXiv:2403.07008, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:34.669899Z"},"links":{"cited_paper":"/paper/2403.07008","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:61ade2016fe31922021005712bddcb3835bc0992b90a4771ff52a1336809c070","observation_id":"3d0e456c-a8c9-4635-8c4b-0b332bff53bd","resolution":{"observed_at":"2026-08-07T12:03:34.669899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:34.895664Z","title":"Adaptively evaluating models with task elicitation.arXiv preprint arXiv:2503.01986, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:34.895664Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:8c8094ad3c20407f2574032e18b2ee460aec321a5c1ce813d477bdb1e8cebcac","observation_id":"e5d3074a-1dc2-4a2c-8ffe-8c5179553f00","resolution":{"observed_at":"2026-08-07T12:03:34.895664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-07T12:03:35.075359Z","title":"Training verifiers to solve math word problems.arXiv preprint arXiv:2110.14168, 2021","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.075359Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:6ec40529e7495eb318a4670d26476579b959470e64f2d40f7ce1bab07512d9eb","observation_id":"b921ed93-8e24-4bca-9e25-0bce68d43405","resolution":{"observed_at":"2026-08-07T12:03:35.075359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:50.502308Z","title":"Knowledge crosswords: Geometric knowledge reasoning with large language models","venue":null,"work_id":"1172f04c-d0f5-4f2c-9bae-241a478933ce","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.188332Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:d1dfa71e648dd1c7f78258474d7e325743019f055d4e0d953bc983af8c260dec","observation_id":"55493653-a93b-4192-b056-e954f47d10a2","resolution":{"observed_at":"2026-08-07T12:03:50.656173Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06961","last_updated":"2024-10-09T14:57:31Z","snapshot_observed_at":"2026-07-06T19:30:26.233621Z","submitted_at":"2024-10-09T14:57:31Z","title":"Self-Boosting Large Language Models with Synthetic Preference Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.06961","snapshot_observed_at":"2026-08-07T12:03:35.341828Z","title":"Self-boosting large language models with synthetic preference data.arXiv preprint arXiv:2410.06961, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.341828Z"},"links":{"cited_paper":"/paper/2410.06961","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:a2d72a8bc9a0905feb0f7c85ce5d045da0ab5428e10310e327df2f5bf7f781c6","observation_id":"d38a2290-f18f-47e1-a916-e0d9b553976b","resolution":{"observed_at":"2026-08-07T12:03:35.341828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:35.502201Z","title":"Alpacafarm: A simulation framework for methods that learn from human feedback.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.502201Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:2d91117662decb5ccff1e7f12abe9c67fd2b3a0c8d776f18e7d03bbce67fea99","observation_id":"013d73e7-0a88-4b2d-b3f3-9d4d03f5362b","resolution":{"observed_at":"2026-08-07T12:03:35.502201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:50.266117Z","title":"Clas- sifying the classifier: dissecting the weight space of neural networks","venue":null,"work_id":"16e7dc4c-7ad7-40b3-86a2-1e458f015daf","year":2020},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.540623Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:aa96bacb4c385d0c84412a786b970bc9438dfdd9c383bcf6e4dd8e91c2b9a70d","observation_id":"d41b1c59-fd0e-4f48-afef-853faba12af0","resolution":{"observed_at":"2026-08-07T12:03:50.381847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:35.613142Z","title":"Heterogeneous swarms: Jointly optimizing model roles and weights for multi-llm systems.arXiv preprint arXiv:2502.04510, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.613142Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:b3fc42b4a93ba259bb2616d055271a2283bbcac816af56cc88ac5bb17635d9d5","observation_id":"e987ecb4-b7f4-4c52-b6c9-3e905dfb908e","resolution":{"observed_at":"2026-08-07T12:03:35.613142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:50.052868Z","title":"Model swarms: Collaborative search to adapt LLM experts via swarm intelligence","venue":null,"work_id":"f3698761-de97-40ea-a6db-8e13dc5c851c","year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.762723Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:4e2c452ba40b0dcf36256921a08d0bc5475b5686236e210fada342907fa1e8d2","observation_id":"fd4fb5fa-c7bd-4b36-85ce-d6907ad443ec","resolution":{"observed_at":"2026-08-07T12:03:50.181618Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:49.781569Z","title":"Promptbreeder: Self-referential self-improvement via prompt evolution","venue":null,"work_id":"b34f4b80-35e2-417c-81f9-3ce9e69ba696","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:35.916500Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:d1947c86304d0b5e0f4eca56177db30f277b64b493a31aedd54f2679189170ca","observation_id":"6301967c-1eb0-41d1-8804-d4b16fd9704a","resolution":{"observed_at":"2026-08-07T12:03:49.915732Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:36.045712Z","title":"Open llm leaderboard v2","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.045712Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:fc1a4b5cf4de3264c3ace3aa8480f84241219a41a2a3d4f8e2f7948850391a15","observation_id":"07f41ea0-74e2-4f30-ae48-8840954b843f","resolution":{"observed_at":"2026-08-07T12:03:36.045712Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:49.527718Z","title":"Time travel in llms: Tracing data contamination in large language models","venue":null,"work_id":"30e0737f-f141-4553-8c7f-05225640c52a","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.184289Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:6880452436b2bf9553fde7e4319ef5f856ee986b7a3795ca57601a6d5a2b7f79","observation_id":"e1da0c0b-27ac-47ea-bb3c-1a6b8142652a","resolution":{"observed_at":"2026-08-07T12:03:49.649629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:49.234081Z","title":"Automated evaluation of retrieval-augmented language models with task-specific exam generation","venue":null,"work_id":"75eed62e-c798-41ef-9661-a8f9d89efa9c","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.332679Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:7c842a43255fe269a73ee4d0cbb309b4409053bf7729fec758f568eb8d4bb22e","observation_id":"8178c412-ca1b-45f4-8c1f-4987319c36c6","resolution":{"observed_at":"2026-08-07T12:03:49.382389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:36.464445Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.464445Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:ca73011777047d61a0ba9d37f756f54eb0d14513fbeb5f3eddac8d2d74bf6c3b","observation_id":"81453eb1-4329-4599-b071-66bf8072cbe2","resolution":{"observed_at":"2026-08-07T12:03:36.464445Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:49.070372Z","title":"Datagen: Unified synthetic dataset generation via large language models","venue":null,"work_id":"5cf03859-9567-40ef-b285-5b29e589ea0c","year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.603676Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:eaaeea26c8c3218d27a872f057da67773eb20ca045048d6dbab31aaf86a513e5","observation_id":"5b49f598-0ae1-451f-a5fa-5464a234093a","resolution":{"observed_at":"2026-08-07T12:03:49.134596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10702","last_updated":"2023-11-20T02:01:33Z","snapshot_observed_at":"2026-07-06T16:49:08.648847Z","submitted_at":"2023-11-17T18:45:45Z","title":"Camels in a Changing Climate: Enhancing LM Adaptation with Tulu 2","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.10702","snapshot_observed_at":"2026-08-07T12:03:36.765030Z","title":"Camels in a changing climate: Enhancing lm adaptation with tulu 2.arXiv preprint arXiv:2311.10702, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.765030Z"},"links":{"cited_paper":"/paper/2311.10702","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:cc81c5dcfe0cd5b53d270b0d31eea8bafd1747832c0eb2607bd3690f988b67e2","observation_id":"4f27019b-e3df-4b9c-8f00-91f772bb388f","resolution":{"observed_at":"2026-08-07T12:03:36.765030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:48.835610Z","title":"Wildteaming at scale: From in- the-wild jailbreaks to (adversarially) safer language models.Advances in Neural Information Processing Systems, 37:47094–47165, 2024","venue":null,"work_id":"3381748b-97a5-430a-8ed6-1ab851b4665e","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:36.978535Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:a4cfa9b9a8c75345a848ce448c0c5dfb286cf5ecff392cc5eb5513fcfb8cc5d5","observation_id":"e55dc512-29eb-492a-8c5c-652e67ef1c96","resolution":{"observed_at":"2026-08-07T12:03:48.937768Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:48.571229Z","title":"Teaching language models to hallucinate less with synthetic tasks","venue":null,"work_id":"ee0dfad2-33b7-453c-825e-64b8b76f85cb","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.116286Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:d768cf9763aa34ae74dc691012987dd496dccfc12145162f2ea3ff921e63bdc4","observation_id":"7002d98f-da27-43be-829b-d8edf4490dcd","resolution":{"observed_at":"2026-08-07T12:03:48.645883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:48.358584Z","title":"Au- tonomous evaluation of llms for truth maintenance and reasoning tasks","venue":null,"work_id":"5ab266ae-e77a-43a7-acdc-fcc1c1cd7e76","year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.251988Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:977da06363c5105d744348d08e9cb5b21e93c5013ebadf6f51b030dbda3c3165","observation_id":"34ed43f9-c088-4ec6-9a3f-ff7c1b7f7e9b","resolution":{"observed_at":"2026-08-07T12:03:48.458711Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:37.403279Z","title":"Realtime qa: What’s the answer right now? Advances in neural information processing systems, 36:49025–49043, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.403279Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:f562a7fd8349329b3a7d5c4e3ea9ce4cb411de62611e175af990fe80e3c221ae","observation_id":"736059c6-e70f-43bd-9a8a-6dc29d8e1b03","resolution":{"observed_at":"2026-08-07T12:03:37.403279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:37.581123Z","title":"Particle swarm optimization","venue":null,"work_id":null,"year":1942},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.581123Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:6c537147e8c5ac762e180508a4b52c1f03b709219e3dfbe07ebf9f7b42689ee2","observation_id":"e30f5e7f-8934-4ad5-8a4c-84496bd00fd0","resolution":{"observed_at":"2026-08-07T12:03:37.581123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:48.105928Z","title":"Openassistant conversations-democratizing large language model alignment.Advances in Neural Information Processing Systems, 36:47669–47681, 2023","venue":null,"work_id":"183826c5-a34e-41c3-83bf-d383c1db38ba","year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.767108Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:425ec8de2988cfd71f65a18ba7b84259407f3027d12f34f827f20ff969ee168b","observation_id":"e7b23351-9ef6-4c55-8752-f02f950bcf67","resolution":{"observed_at":"2026-08-07T12:03:48.194529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01236","last_updated":"2025-02-03T10:52:44Z","snapshot_observed_at":"2026-07-06T20:30:12.663808Z","submitted_at":"2025-02-03T10:52:44Z","title":"Eliciting Language Model Behaviors with Investigator Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01236","snapshot_observed_at":"2026-08-07T12:03:37.921485Z","title":"Eliciting language model behaviors with investigator agents.arXiv preprint arXiv:2502.01236, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:37.921485Z"},"links":{"cited_paper":"/paper/2502.01236","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:3e59a73c9c940368c1585d882de9d48e0dccaad36639dba37b4e1c842fa19918","observation_id":"e40cc949-e6b2-43c0-b393-e6d8927bf100","resolution":{"observed_at":"2026-08-07T12:03:37.921485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:38.098878Z","title":"Autobencher: Towards declarative benchmark construction","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:38.098878Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:cb95e85f60edd7306def0539666309ebfa526f03d139e1a6f830c275fc39a889","observation_id":"f3b21320-0d42-4975-ba23-a63db57a5890","resolution":{"observed_at":"2026-08-07T12:03:38.098878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:47.868331Z","title":"Gen- dataagent: On-the-fly dataset augmentation with synthetic data","venue":null,"work_id":"7529d163-3ebc-44be-9507-dc63691a72e7","year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:38.227102Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:053d4986c360c59e2ed7d7bd1f75515a9d2426e4fd80d3d57eb968598a45f890","observation_id":"d244d9dd-3773-4826-ad02-5471a91be604","resolution":{"observed_at":"2026-08-07T12:03:47.979881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:47.525412Z","title":"Hemm: Holistic evaluation of multimodal foundation models","venue":null,"work_id":"7b6c212d-fc82-49c5-9d01-0961d9e99d80","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:38.373411Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:666a951839dfda9554ed7705e5388fb7039057d8b2fbe5f9742088ee671960ee","observation_id":"8da932b7-31ef-4c50-8c38-e62277f8c458","resolution":{"observed_at":"2026-08-07T12:03:47.710055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:47.321917Z","title":"Holistic evaluation of language models.Transactions on Machine Learning Research, 2022","venue":null,"work_id":"be0bbbc0-b3ab-423b-8967-1d194ec76707","year":2022},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:38.517906Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:797b085af1eabe6f0f56a8e55eb63c82912b92fb00c63591db3339fcfe16a450","observation_id":"a89be644-4aab-44c0-92d4-34e813ff1c33","resolution":{"observed_at":"2026-08-07T12:03:47.397511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:38.666791Z","title":"Truthfulqa: Measuring how models mimic human falsehoods","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:38.666791Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:9b1e4ed0d85de96071c687cfb2b1c9afbe5ffcb387a05091f6b5cc86b66b0ef9","observation_id":"50e6751b-e4e9-4fd8-b1ca-b844c8542f33","resolution":{"observed_at":"2026-08-07T12:03:38.666791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:47.070962Z","title":"Best practices and lessons learned on synthetic data","venue":null,"work_id":"9f724313-aa7d-4312-aa56-225a265f5773","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:38.854417Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:9f846084ec9c7bbe8e8ca00643bdbd651cab597df5888224e46909f8dfa51053","observation_id":"34406dc8-cbb5-4a5b-a9e9-46bc8580c585","resolution":{"observed_at":"2026-08-07T12:03:47.205529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.19290","last_updated":"2024-10-25T03:48:51Z","snapshot_observed_at":"2026-07-06T19:39:28.707989Z","submitted_at":"2024-10-25T03:48:51Z","title":"Fictitious Synthetic Data Can Improve LLM Factuality via Prerequisite Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.19290","snapshot_observed_at":"2026-08-07T12:03:39.014964Z","title":"Fictitious synthetic data can improve llm factuality via prerequisite learning.arXiv preprint arXiv:2410.19290, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.014964Z"},"links":{"cited_paper":"/paper/2410.19290","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:2e71af83e2257885382ab08f5435b5c985fbeaefc5504e094e63eaff89064ae6","observation_id":"080387ab-ccae-423b-adc7-7451384d0e5d","resolution":{"observed_at":"2026-08-07T12:03:39.014964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:46.575965Z","title":"Adaptive labeling for efficient out-of-distribution model evaluation.Advances in Neural Information Processing Systems, 37:70981–71003, 2024","venue":null,"work_id":"1a6b4fee-5517-4f2e-bd95-76f7a2bad68c","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.157235Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:3371bb5f603c1b34ad9a08e3508d275ae8f5279fb71c4141a58d42226ab10129","observation_id":"ec6e65cc-1c8c-4b78-b252-8c1ea0f7441f","resolution":{"observed_at":"2026-08-07T12:03:46.853533Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13826","last_updated":"2024-10-24T17:27:22Z","snapshot_observed_at":"2026-07-06T19:35:30.732080Z","submitted_at":"2024-10-17T17:51:40Z","title":"Unearthing Skill-Level Insights for Understanding Trade-Offs of Foundation Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13826","snapshot_observed_at":"2026-08-07T12:03:39.313256Z","title":"Unearthing skill-level insights for understanding trade-offs of foundation models.arXiv preprint arXiv:2410.13826, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.313256Z"},"links":{"cited_paper":"/paper/2410.13826","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:c146a9c13858cb3bcf4ebffc87adf6cfcc754e397afa7c87552862968dde257a","observation_id":"93bb72f4-c315-4cbb-9533-827ac636cf70","resolution":{"observed_at":"2026-08-07T12:03:39.313256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:46.322482Z","title":"Enhancing reason- ing capabilities of llms via principled synthetic logic corpus.Advances in Neural Information Processing Systems, 37:73572–73604, 2024","venue":null,"work_id":"f8bd2a5c-e246-45d7-a3b2-51d8cad2fafc","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.451503Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:a9d5668afc28c040726ba1c1e541cc5527530112fe035d05f5201877d385a700","observation_id":"4c594c94-5c1a-42e8-8e81-dcd7a06e0923","resolution":{"observed_at":"2026-08-07T12:03:46.425652Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:46.140469Z","title":"Caps: Collaborative and private synthetic data generation from distributed sources","venue":null,"work_id":"f623fe16-7d6a-4f9c-8d45-d9bc74e9caf5","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.570494Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:c5cffa614cfc4385fc3a1ac1803348ae9628bb7a5e10e19471d5b141bf840ba2","observation_id":"68588ff6-2b66-43e7-8058-4dc7072c5046","resolution":{"observed_at":"2026-08-07T12:03:46.206363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-07T12:03:39.704349Z","title":"Humanity’s last exam.arXiv preprint arXiv:2501.14249, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.704349Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:6c6ee6e1768f6e4fc184a0e6859aae51a3a32759ca3e85e32bf4031367486fd0","observation_id":"c185db6c-5de2-4728-b983-20224c9e5697","resolution":{"observed_at":"2026-08-07T12:03:39.704349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:39.843361Z","title":"Quantifying language models’ sensitivity to spurious features in prompt design or: How i learned to start worrying about prompt formatting","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.843361Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:b275e26669bce99d562b1869e91a7d91c7873f72cef4d521ca350f282ef9413c","observation_id":"1440dcac-f8d7-4683-9148-2dba3f03f29a","resolution":{"observed_at":"2026-08-07T12:03:39.843361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12175","last_updated":"2024-12-12T21:29:00Z","snapshot_observed_at":"2026-07-06T20:08:02.963967Z","submitted_at":"2024-12-12T21:29:00Z","title":"Explore Theory of Mind: Program-guided adversarial data generation for theory of mind reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12175","snapshot_observed_at":"2026-08-07T12:03:39.949056Z","title":"Explore theory of mind: Program-guided adversarial data generation for theory of mind reasoning.arXiv preprint arXiv:2412.12175, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:39.949056Z"},"links":{"cited_paper":"/paper/2412.12175","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:6625904d224eae5a813b505b2bf28e29fea7f765cca2febffb49bcf1bbe6d1b4","observation_id":"4b304ac3-1d89-486f-a811-dd6974a27604","resolution":{"observed_at":"2026-08-07T12:03:39.949056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:40.030301Z","title":"Rl on incorrect synthetic data scales the efficiency of llm math reasoning by eight-fold.Advances in Neural Information Processing Systems, 37:43000–43031, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.030301Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:9a4536f2d716d1a74a0a5f494067dce98ba2f51bb2b17e79d4df7bf805a00f04","observation_id":"a82e3d99-1f46-439c-93e6-25414521ab33","resolution":{"observed_at":"2026-08-07T12:03:40.030301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.16342","last_updated":"2025-02-19T03:56:29Z","snapshot_observed_at":"2026-08-03T22:27:41.536880Z","submitted_at":"2024-06-24T06:27:47Z","title":"Is your benchmark truly adversarial? AdvScore: Evaluating Human-Grounded Adversarialness","version":3},"cited_work":{"arxiv_id":"2406.16342","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.16342","snapshot_observed_at":"2026-08-07T12:03:42.886247Z","title":"Is your benchmark truly adversarial? AdvScore: Evaluating Human-Grounded Adversarialness","venue":"cs.CL","work_id":"af48a0d1-2973-42ea-833d-8ec1ad94954b","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.130964Z"},"links":{"cited_paper":"/paper/2406.16342","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:f2e6bd6433c9b94c99b9303f22555ae51a75462dc7e17af0bbef6b41692c32c1","observation_id":"2d635b6f-aa87-4f42-b0eb-f251f2f356b5","resolution":{"observed_at":"2026-08-07T12:03:42.954865Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-07T12:03:40.248109Z","title":"Gemma 2: Improving open language models at a practical size.arXiv preprint arXiv:2408.00118, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.248109Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:417b3110d3e27c08dd318c3dad9891134ae07b973d6936b97fdb365bf4e065cc","observation_id":"a4ceafbe-6bbf-488b-b76a-aeffe7a8c783","resolution":{"observed_at":"2026-08-07T12:03:40.248109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:40.350204Z","title":"Measuring general intelligence with generated games, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.350204Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:528af3f54956554fe1d1dd947eafcd5eebe77d53ebea533a847f19388f7211a0","observation_id":"83c22784-02cf-4af9-8e02-6683f78e080e","resolution":{"observed_at":"2026-08-07T12:03:40.350204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:45.922113Z","title":"Cuts: Customizable tabular synthetic data generation","venue":null,"work_id":"3315dcdf-5112-4d81-b371-dc676ad7eab7","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.455814Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:1b9b8fe5cd7886e06b1b9f3bad81da4218fcd915473458c4f78ad13c5e7d3f7a","observation_id":"287aab16-f8da-4da7-aaa2-a1a53c8a67fe","resolution":{"observed_at":"2026-08-07T12:03:45.994917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12480","last_updated":"2025-03-12T22:04:34Z","snapshot_observed_at":"2026-07-06T18:32:53.080027Z","submitted_at":"2024-06-18T10:36:21Z","title":"The Power of LLM-Generated Synthetic Data for Stance Detection in Online Political Discussions","version":2},"cited_work":{"arxiv_id":"2406.12480","doi":null,"metadata_source":"pith","pith_arxiv_id":"2406.12480","snapshot_observed_at":"2026-08-07T12:03:42.655300Z","title":"The Power of LLM-Generated Synthetic Data for Stance Detection in Online Political Discussions","venue":"cs.CL","work_id":"f84793ce-d602-4156-891a-aad55531759b","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.514127Z"},"links":{"cited_paper":"/paper/2406.12480","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:5ad3989de03d50c952e88bb6cc5a48afa5c7e05533f76bc091e113db21594941","observation_id":"fa9c1b34-5dae-480e-8558-4dea15fe50b3","resolution":{"observed_at":"2026-08-07T12:03:42.726856Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:45.715038Z","title":"Glue: A multi-task benchmark and analysis platform for natural language understanding","venue":null,"work_id":"84ffd11b-c6b2-45bf-ae8a-da9a476a9373","year":2019},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.582255Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:dc00324926d4b0fdd64ce84ede979737b9d67f0164a4608784994218d69e0eea","observation_id":"12ecc956-b073-4191-9415-7a6d530c350f","resolution":{"observed_at":"2026-08-07T12:03:45.805852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:45.518057Z","title":"Can language models solve graph problems in natural language?Advances in Neural Information Processing Systems, 36:30840–30861, 2023","venue":null,"work_id":"9e30882f-a742-4022-a453-45536a3c5e8b","year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.660003Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:55755116974dd561856975c611e4c4cff59415363578bcad7c6ca604a7575b90","observation_id":"e5220f4a-7129-497a-b745-4f8170da5201","resolution":{"observed_at":"2026-08-07T12:03:45.621047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:45.302277Z","title":"Pandalm: An automatic evaluation benchmark for llm instruction tuning optimization","venue":null,"work_id":"4a5264a5-4872-473e-937f-5e004332f670","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.729147Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:19d22aedd9d1d4cb521c1a0d216a784e19a705fc7e17041e1d3d5c6047873b8a","observation_id":"b7cc40a9-d296-4b25-ac97-2c099f57c4c4","resolution":{"observed_at":"2026-08-07T12:03:45.358576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:40.827284Z","title":"Self-instruct: Aligning language models with self-generated in- structions","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.827284Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:173ce9464f1a626bae7b52b0153ba9dde2aa1167d279ce5a5f31cd50de3ba807","observation_id":"1064b364-68fa-45f1-9753-a3f9176832e9","resolution":{"observed_at":"2026-08-07T12:03:40.827284Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:45.136613Z","title":"Pre-training with synthetic data helps offline reinforcement learning","venue":null,"work_id":"454c2398-1d63-4c80-a9c3-9db319d1705b","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.925782Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:81c38437029a868b1eee995e6959339f4c770006c26ff5990afff253766d08c5","observation_id":"f672e44b-eda7-4f67-8007-da3b6ac579dc","resolution":{"observed_at":"2026-08-07T12:03:45.214965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:44.932827Z","title":"Rocketeval: Efficient automated llm evaluation via grading checklist","venue":null,"work_id":"08197fc7-94a7-4a05-9734-22f6a069670e","year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:40.994839Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:830bd5147c8e66252f59c2fbd9306962d9cb9c394c4acdbe8a69fc207c2e1485","observation_id":"2ae4d87d-0c52-43f4-8558-51801c4ca646","resolution":{"observed_at":"2026-08-07T12:03:45.017392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:41.104348Z","title":"Model soups: averaging weights of multiple fine-tuned models improves accuracy without increasing inference time","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.104348Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:b1628699a62d90c3615852d14111f2bd0c059cf6db5be40fd1b4d635e7435d64","observation_id":"a5a81aa4-3368-492c-8936-94765473778d","resolution":{"observed_at":"2026-08-07T12:03:41.104348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:44.740659Z","title":"Differentially private synthetic data via foundation model apis 2: Text","venue":null,"work_id":"783165a5-5099-425f-b550-920f2df4453c","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.220080Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:8906a688957dccb732bb7cee737f19264765bdaa1e8888079f570d143e2593cb","observation_id":"5a54daff-d92d-42da-95e8-c21d034ee6d0","resolution":{"observed_at":"2026-08-07T12:03:44.837302Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:44.579212Z","title":"Automating dataset updates towards reliable and timely evaluation of large language models","venue":null,"work_id":"0de4cadf-9f54-4e2e-940f-a1a7f25d25a4","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.331758Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:cdb185e6c7973f21700c68a10d77c46d047bcd0bc94844d8fd2b9872a728af55","observation_id":"d570d6b4-12d0-49a0-b1d7-6ca0dab74574","resolution":{"observed_at":"2026-08-07T12:03:44.651043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:44.399571Z","title":"xfinder: Large language models as automated evaluators for reliable evaluation","venue":null,"work_id":"9ebbe822-bacb-4c91-9b5f-6a6969429780","year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.402978Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:b1ead2f404a7b59c690f00fec692380b1791a87545b2bd7ceb04c229b31ed685","observation_id":"22db04ed-bee3-4fd8-a2eb-5c6de61b3b6e","resolution":{"observed_at":"2026-08-07T12:03:44.481634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:41.541623Z","title":"Hellaswag: Can a machine really finish your sentence? InProceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pages 4791–4800, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.541623Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:ec3f6340e69f25adcd3f96fc8505bce053143d6da16bb122659e5e61c9c66f9c","observation_id":"36e675ce-81da-4eaf-a022-33e188051eeb","resolution":{"observed_at":"2026-08-07T12:03:41.541623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.08893","last_updated":"2025-07-11T05:27:37Z","snapshot_observed_at":"2026-08-07T17:11:45.488478Z","submitted_at":"2025-03-11T21:12:48Z","title":"EvalTree: Profiling Language Model Weaknesses via Hierarchical Capability Trees","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.08893","snapshot_observed_at":"2026-08-07T12:03:41.647357Z","title":"Evaltree: Profiling language model weaknesses via hierarchical capability trees.arXiv preprint arXiv:2503.08893, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.647357Z"},"links":{"cited_paper":"/paper/2503.08893","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:1637cd44616207c5fcd04b1b79d1bcf134ecd8281ad5283784d7448d471202ff","observation_id":"e5df9015-eeba-493d-a863-5c1d497aa6a3","resolution":{"observed_at":"2026-08-07T12:03:41.647357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.12306","last_updated":"2025-05-18T08:39:05Z","snapshot_observed_at":"2026-08-07T15:43:43.434824Z","submitted_at":"2025-05-18T08:39:05Z","title":"Bidirectional LMs are Better Knowledge Memorizers? A Benchmark for Real-world Knowledge Injection","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.12306","snapshot_observed_at":"2026-08-07T12:03:41.749831Z","title":"Bidirectional lms are better knowledge memorizers? a benchmark for real-world knowledge injection.arXiv preprint arXiv:2505.12306, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.749831Z"},"links":{"cited_paper":"/paper/2505.12306","citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:7fd1e51db040d378ad0df2d2b56314c2723e717945faaf6536934d7dc949017f","observation_id":"421897cc-4d5d-4cfe-90cd-ce34ec6505a1","resolution":{"observed_at":"2026-08-07T12:03:41.749831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:44.254697Z","title":"Wildchat: 1m chatgpt interaction logs in the wild","venue":null,"work_id":"c93b71c7-6d0f-4538-a867-ac7c65445d83","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.852941Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:2bcce9353edae3cb8c2028b38df1acd69bf6cc0a8ed10ae1c2dd95881dd50015","observation_id":"7fff0ab4-4543-462f-9808-82a749400c7f","resolution":{"observed_at":"2026-08-07T12:03:44.304019Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:41.967706Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena.Advances in Neural Information Processing Systems, 36:46595–46623, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:41.967706Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:65f070b851f8ffecb34d64d8be83fa5e41d5039395bda3c97f6d18252f428071","observation_id":"523b8033-85f6-4e45-b746-d0638c7ad679","resolution":{"observed_at":"2026-08-07T12:03:41.967706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:44.078023Z","title":"Lima: Less is more for alignment.Advances in Neural Information Processing Systems, 36:55006–55021, 2023","venue":null,"work_id":"1907ac09-d9d5-4174-a84c-2dedd61e1735","year":2023},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:42.068582Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:8a0821c97cfcf83939df75535d90e073b5f105f582c9364d79f54a410252f23c","observation_id":"c2b40608-3786-4bce-bc64-3ffb03f02354","resolution":{"observed_at":"2026-08-07T12:03:44.147466Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:43.907919Z","title":"Sotopia: Interactive evaluation for social intelligence in language agents","venue":null,"work_id":"ab6d2d53-bc43-46fb-8e02-9dab1201d834","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:42.161531Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:2c531bb3188fb03a3a6ea6228dad5d68df87811253faefb8283b18a3e64d5b02","observation_id":"3f59b8ae-51c8-4f18-bccd-539b6abfc1b2","resolution":{"observed_at":"2026-08-07T12:03:43.981981Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:43.735463Z","title":"Dyval: Dynamic evaluation of large language models for reasoning tasks","venue":null,"work_id":"32fca5e4-4d44-4cd3-9425-f1815ffad701","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:42.262166Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:d3bc8600559c2b6b2e64378c7583083db0a93560d80303f99b57941b659d51c6","observation_id":"49d8ee22-76eb-4750-a158-8d9a5de1bc00","resolution":{"observed_at":"2026-08-07T12:03:43.812704Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:43.571296Z","title":"Dynamic evaluation of large language models by meta probing agents","venue":null,"work_id":"436f9bea-733d-4d60-8bef-6ce1e6e97672","year":2024},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:42.372390Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:b4689b39b9d509ee4402ef386fe85104cc2d518cd4e1c67874da54277c345255","observation_id":"170741c2-449c-4e06-b467-adda07e6e7b4","resolution":{"observed_at":"2026-08-07T12:03:43.645456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T12:03:43.422190Z","title":"Top Secret","venue":null,"work_id":"1e1ac16b-f41c-4838-ac77-79ee3b2ead17","year":1953},"citing_paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T12:03:42.484850Z"},"links":{"citing_paper":"/paper/2506.00741"},"observation_digest":"sha256:57320d79087103ab95ea24dfc5934150c69d4a926477c46f652fc7188bcd072b","observation_id":"09208314-afdc-414f-8d4a-dc9ae8757df8","resolution":{"observed_at":"2026-08-07T12:03:43.487172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.00741","last_updated":"2025-06-06T02:20:24Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T11:56:44.689446Z","submitted_at":"2025-05-31T23:03:46Z","title":"Data Swarms: Optimizable Generation of Synthetic Evaluation Data"},"reference_resolution":{"displayed":64,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":2,"verified_fuzzy":33},"total_outbound_references":64},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 64 of 64 outbound references and 0 inbound Pith citation observations for arXiv:2506.00741."}