{"as_of":"2026-08-07T19:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f6d25db44e789d2311e94629ceaae9f470422f027bb99e75abfdca4272a9b02c","coverage":[{"denominator":68,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":68,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:43:42.163889Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.05266/citation-record","integrity":"/paper/2507.05266/integrity","json":"/paper/2507.05266/citation-record.json","paper":"/paper/2507.05266"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:31.780206Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:31.780206Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:fdd3f0527ce498756c6b0512233d40dee9e58111a1ef9246dc1916c78ff0fbbe","observation_id":"302b938b-9e5d-4e70-ab47-d0fb66818a15","resolution":{"observed_at":"2026-08-06T21:43:31.780206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T21:43:31.875312Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:31.875312Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:d1419d38b8e937977940d7973517bca53ac6451d3d8224efa8ce49965ab2affe","observation_id":"5a840d97-3a01-4e97-833e-d55072cc25eb","resolution":{"observed_at":"2026-08-06T21:43:31.875312Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:31.999784Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:31.999784Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:78110fe85999a3471e1a0af90e2a7bea633177b4eedd4d92a5add15b5c1f5900","observation_id":"57efcae8-9538-4c16-a50f-102830ffb2a3","resolution":{"observed_at":"2026-08-06T21:43:31.999784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:32.114826Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:32.114826Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a527237bb27dbcb94d775960c0a89e29cc60584cb88cf2f0aa66eca34dedaa6a","observation_id":"2dc54533-f717-4608-b31b-3dba1fde6e29","resolution":{"observed_at":"2026-08-06T21:43:32.114826Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:52.296754Z","title":null,"venue":null,"work_id":"38b56e00-cce0-4b1c-a6e5-0047dfe0e627","year":2005},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:32.258580Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:788d18242008b66b1cf199222b9262372bebcdc388a2df0e6241054d5dfd24a1","observation_id":"10879fc8-cf60-4d2e-a67f-b9dcbfe83cac","resolution":{"observed_at":"2026-08-06T21:43:52.458598Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:51.973614Z","title":null,"venue":null,"work_id":"a3f0822f-b7ca-46ce-b033-6f3e798a00c1","year":2006},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:32.376986Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:f6ee987e85d0e9f54e56d6edaf8996e793a05241ec7bb4a064530af2c2b70528","observation_id":"8974b10e-f782-4c6d-9f3f-1cb23bccc5fd","resolution":{"observed_at":"2026-08-06T21:43:52.151476Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:32.573252Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:32.573252Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:1714380a4d2a42a9dbc0d12893a1c8d4297a0ad9916d1e3bac567731cc5d6e20","observation_id":"6e57049b-4de5-4a73-b7ed-a5192107b069","resolution":{"observed_at":"2026-08-06T21:43:32.573252Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:51.699585Z","title":null,"venue":null,"work_id":"2caacbb1-f5a6-472d-8e0f-95641a2ee26e","year":2017},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:32.738990Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:639ea8b488716d57aff478861eb12691b04a8226a4c27bae4d8e225e608024ee","observation_id":"01c74ad5-effb-4355-9775-7a999a9e6d33","resolution":{"observed_at":"2026-08-06T21:43:51.843815Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:32.886441Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:32.886441Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:6478642baec5dc6ae23e79fe1951bf18145d8a404e1236ff492a57e861eccdc8","observation_id":"d8d53ded-173b-47a0-a3e6-520eb5fb0c35","resolution":{"observed_at":"2026-08-06T21:43:32.886441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:51.416262Z","title":null,"venue":null,"work_id":"d998fad8-9db7-4d4b-9261-0e6b8669184e","year":2010},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.047126Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:f699eec218df786cb8709a7d76fba3214e45e4b0d51bcbb646266eeac2eddbb2","observation_id":"3f64f436-4568-43ac-aecc-20b0dc93b1a0","resolution":{"observed_at":"2026-08-06T21:43:51.533764Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:33.188732Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.188732Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:b3e8de83bb28fe8b9fa07f5a185835601302ffe59b73ab19e17c18e229dbfbed","observation_id":"66a0cc6b-f77d-4826-9a2e-7b760a2a1bd6","resolution":{"observed_at":"2026-08-06T21:43:33.188732Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:51.115765Z","title":null,"venue":null,"work_id":"453410d7-7eba-4ea5-a92c-e2ca560f0c95","year":2010},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.390981Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:02934f754091ac7405d874fe52d03ac451178cd02db25a95a4a6c9436fd8dcea","observation_id":"1eae7162-38e5-47b4-9009-0d5023475ea6","resolution":{"observed_at":"2026-08-06T21:43:51.250174Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17161","last_updated":"2025-05-26T17:16:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-28T18:59:44Z","title":"SFT Memorizes, RL Generalizes: A Comparative Study of Foundation Model Post-training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17161","snapshot_observed_at":"2026-08-06T21:43:33.539108Z","title":"Le, Sergey Levine, and Yi Ma","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.539108Z"},"links":{"cited_paper":"/paper/2501.17161","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:3663aa762b26207e7e99e1ca11389bc3aa9c9888deed1908a7851c18b34e1e53","observation_id":"fa79763b-22ed-41b7-a961-63e28d4b1669","resolution":{"observed_at":"2026-08-06T21:43:33.539108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-07T01:45:38.840969Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-06T21:43:33.704744Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.704744Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:9cd67ada398d222940df3a2baeea2f12f7f64df9e7bde1eba88b40451662c8cf","observation_id":"bbe85ef2-aba5-4247-83d3-15b00f701a1b","resolution":{"observed_at":"2026-08-06T21:43:33.704744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:50.819634Z","title":null,"venue":null,"work_id":"ebf4ac0c-f5a1-48af-94fe-2f8c41e70ef5","year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.830889Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:97dad8db3897f10d82148370c4829c58cdca678f6eb9e2701ecb609e38d373a8","observation_id":"705695e9-9f3a-4a10-aeff-888991ad8aa7","resolution":{"observed_at":"2026-08-06T21:43:50.952985Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:50.533520Z","title":null,"venue":null,"work_id":"520b95a5-cb7f-4a1a-87cf-9d0827678652","year":2000},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:33.956895Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:6c7f141f89dfeca9f93d56cbfa156331e6924c72f9bd483b7c27418faac03312","observation_id":"95ad60b4-0b66-483d-9a5e-4501433dc73b","resolution":{"observed_at":"2026-08-06T21:43:50.673809Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:50.337954Z","title":null,"venue":null,"work_id":"ab308944-8617-4f66-88df-77a8069b1bbf","year":2006},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:34.096827Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:458203cceffdceb3c6ffca8ec58738940255dbd5b808ea4611e92ac64c21a2ed","observation_id":"3b0cb5c1-c708-4a80-b7be-469ea929e3ee","resolution":{"observed_at":"2026-08-06T21:43:50.419361Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T21:43:34.302368Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:34.302368Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:b158e0f612301664769874d43aa85b79932f7ed1e2865e40db3d696d917893b1","observation_id":"807f6f38-74f8-48b6-9c23-fbd04f1143d1","resolution":{"observed_at":"2026-08-06T21:43:34.302368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:50.141929Z","title":null,"venue":null,"work_id":"87c002cf-8b45-48f2-a410-6e51cddf5eb7","year":2006},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:34.460806Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:1606ef573c1d652cab7af303f68cbaf68feffcb27f59634f015a10acfc220d2e","observation_id":"4c2fe7a5-9d42-48a0-a81b-7f7164fb809e","resolution":{"observed_at":"2026-08-06T21:43:50.242081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:49.866805Z","title":null,"venue":null,"work_id":"2a869e40-ac21-41f6-bd19-cae85bf50cfe","year":2016},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:34.645032Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:923628c57e9142c4c5ab23497d9a3254086d45d57e161bc3482d9d3d68af849c","observation_id":"c3602873-a393-4ebd-9cca-f3786a0a1db6","resolution":{"observed_at":"2026-08-06T21:43:50.049174Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:49.569672Z","title":null,"venue":null,"work_id":"1fd13edd-6f2c-47c1-9473-7c5bdd71bacf","year":2020},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:34.815490Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:b1e8c651d7f9ab4cfda0ef801b85467044290f702c60a61e4e3c1b50630a92cb","observation_id":"678f20e6-1941-4621-8df4-ba5e34ad7027","resolution":{"observed_at":"2026-08-06T21:43:49.698061Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19736","last_updated":"2023-11-25T17:35:12Z","snapshot_observed_at":"2026-07-06T16:40:35.590974Z","submitted_at":"2023-10-30T17:00:52Z","title":"Evaluating Large Language Models: A Comprehensive Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19736","snapshot_observed_at":"2026-08-06T21:43:34.962660Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:34.962660Z"},"links":{"cited_paper":"/paper/2310.19736","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:38a340ab433f3c38ef86ac90cacfc9f9dd4f0b34ca7d72b18a487077943f79fa","observation_id":"528253e7-3801-47df-a54b-7e0035840e8d","resolution":{"observed_at":"2026-08-06T21:43:34.962660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:49.250802Z","title":null,"venue":null,"work_id":"e59e47bd-09a6-4bf6-a2eb-a54cea05f406","year":2006},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:35.108577Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:ef01be0989facf460249c909993a5de109da4f6b0242b0e29b0c8c7de99e77ec","observation_id":"1c48d83d-abd6-424c-8494-15464844d20c","resolution":{"observed_at":"2026-08-06T21:43:49.432484Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:49.043599Z","title":null,"venue":null,"work_id":"6aa4a8e4-1890-4287-bbdb-08cce6c22c69","year":2015},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:35.348888Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:71d1864c16eb74a8447035c7ac92ddf2d930cc0dab6e4d04ef1a9f681bee6f23","observation_id":"cbcdbf1d-310b-4a80-a839-33bc7136cb65","resolution":{"observed_at":"2026-08-06T21:43:49.145384Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:35.518142Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:35.518142Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:2c1715677d0f44497042716d090be13674d8b3352f3c7dfb1de0262890606c82","observation_id":"fc879cd2-5b81-46d5-99b3-f9ba5aaac410","resolution":{"observed_at":"2026-08-06T21:43:35.518142Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-06T21:43:35.760263Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:35.760263Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:e92651aee5e51bf187b4a6d258f8d06d5f35399dc114ca84bf592876fe5f8003","observation_id":"84fe2f46-16c0-4b57-8df9-f2cfeea8c4ab","resolution":{"observed_at":"2026-08-06T21:43:35.760263Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:48.716166Z","title":null,"venue":null,"work_id":"f4208ea6-365e-412b-b3db-7dc11ccf97c2","year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:35.962034Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:d8a88899a940225ca967f4e57816d6d94effa05c9b010b2d55e4103c4e77720e","observation_id":"f1e1386a-ead7-4b1d-8ad4-0dbe18ca923b","resolution":{"observed_at":"2026-08-06T21:43:48.855108Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:36.088502Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:36.088502Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:de7cfee315759fa7b4a825d8524eec241669d4c98a180f216abc05b301f031b4","observation_id":"77459335-f923-4f6a-8c3a-38edff5cbd33","resolution":{"observed_at":"2026-08-06T21:43:36.088502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:48.477240Z","title":null,"venue":null,"work_id":"815d383e-b0d0-407e-8546-404b77124480","year":2002},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:36.226337Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a5a0dbd9a0d152685b88f3acd93968a29610505cf041d8667af079d3353a6efd","observation_id":"d5079bbc-8f53-448a-adf6-47548034f98f","resolution":{"observed_at":"2026-08-06T21:43:48.554427Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:48.210359Z","title":null,"venue":null,"work_id":"4f6e787c-90a3-4ea8-9305-c1189c28d773","year":2022},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:36.406874Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:156294e55a254bc84c6d63d27c2b2a19fa5c37cfeeaf035a83d1ce30effffe22","observation_id":"2587c13a-aa98-4a9e-9db5-acafcc25eea8","resolution":{"observed_at":"2026-08-06T21:43:48.313627Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.07870","last_updated":"2023-11-07T16:28:33Z","snapshot_observed_at":"2026-07-06T15:54:25.070719Z","submitted_at":"2023-07-15T19:04:33Z","title":"Large Language Models as Superpositions of Cultural Perspectives","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.07870","snapshot_observed_at":"2026-08-06T21:43:36.579640Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:36.579640Z"},"links":{"cited_paper":"/paper/2307.07870","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:031510996516e0487529ba1c0fb5772748631f4061d169582b477301100ede7f","observation_id":"7bdf0408-6c0d-48bf-977e-d323094bd47b","resolution":{"observed_at":"2026-08-06T21:43:36.579640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:47.997518Z","title":null,"venue":null,"work_id":"b2a8f2f1-5b94-476f-a029-93c13ec146e7","year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:36.780484Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a60ad83dff6b64113898b49d66f8d80a75b2ee1607f89c530bc84102a19258d1","observation_id":"80cc9615-efe8-448c-95d2-4e274449483e","resolution":{"observed_at":"2026-08-06T21:43:48.069470Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:47.769743Z","title":null,"venue":null,"work_id":"12f788ec-1476-46ef-a282-e86def8c8762","year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:36.953376Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a8689605446611069c0799fa4ed04094dbb88bffb0fdce618f25b362a64a44fc","observation_id":"11eeba2d-8f91-4911-a173-b4a508117b1a","resolution":{"observed_at":"2026-08-06T21:43:47.902732Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.10199","last_updated":"2024-08-20T06:53:45Z","snapshot_observed_at":"2026-07-06T18:00:39.863947Z","submitted_at":"2024-04-16T00:50:43Z","title":"CULTURE-GEN: Revealing Global Cultural Perception in Language Models through Natural Language Prompting","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.10199","snapshot_observed_at":"2026-08-06T21:43:37.167104Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:37.167104Z"},"links":{"cited_paper":"/paper/2404.10199","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:e5eab25579ebce182fb7962c797f5e6bfb3efa4801d37090fe64a063b14dfffa","observation_id":"465c50b7-af95-4c84-9083-f0f3c6ad6568","resolution":{"observed_at":"2026-08-06T21:43:37.167104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:47.497963Z","title":null,"venue":null,"work_id":"f2cefcdb-4525-4212-8532-a44daf01ab47","year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:37.358895Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:de9caf4363ae48ce4fe2524194bc1744b9dbbddfdd4777b2b01f5557f4727057","observation_id":"f1d3606a-65c6-4009-a9f1-367353ba9ead","resolution":{"observed_at":"2026-08-06T21:43:47.615443Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10149","last_updated":"2023-10-27T09:11:50Z","snapshot_observed_at":"2026-07-06T15:17:48.587665Z","submitted_at":"2023-04-20T08:16:07Z","title":"Is ChatGPT a Good Recommender? A Preliminary Study","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10149","snapshot_observed_at":"2026-08-06T21:43:37.525608Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:37.525608Z"},"links":{"cited_paper":"/paper/2304.10149","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:b5698b0794d6f83cb2325578052e87b408855bb0784e47e78fb8d3a63d4ba56d","observation_id":"6a3eae84-0260-4fd0-821a-3623203ddb66","resolution":{"observed_at":"2026-08-06T21:43:37.525608Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:47.290557Z","title":null,"venue":null,"work_id":"141e4da3-6b78-4a41-a52b-6ba8dfffb9ca","year":2019},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:37.644108Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:1837f4f951decbc6b68c6a243f13a8efbb8bfa745c09938ff1ffc0711eef06fe","observation_id":"d7b35eae-0f60-46cc-a997-1c1aae56d2f0","resolution":{"observed_at":"2026-08-06T21:43:47.369504Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.884","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:42.309601Z","title":null,"venue":null,"work_id":"f64a9b15-3b80-4dc1-9ea9-914e8540ecdf","year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:37.764807Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:f36bb91c64f70503fd88b50946212c2d9e6dabf3fec30295553a428be54d1bc4","observation_id":"e1e1076a-28ec-41cb-ba97-26630c9cbda7","resolution":{"observed_at":"2026-08-06T21:43:42.403995Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:37.901202Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:37.901202Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:9fc1546bcea4ba5688ff285b9ff7f5dee368ef9e76663d96356d59445136ada7","observation_id":"ee09ef14-54c4-46d8-889d-3c3f51247fca","resolution":{"observed_at":"2026-08-06T21:43:37.901202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:47.031021Z","title":null,"venue":null,"work_id":"b89e6aa0-d337-4da9-87e0-72fa7828084f","year":2015},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:38.022415Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a9f23824288fa1474d25cdd9f506504dd525c760b88e5ae36dbf8ce3e20dc942","observation_id":"dd3faa59-e19b-491d-8af2-570c6bc63660","resolution":{"observed_at":"2026-08-06T21:43:47.152472Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:38.257084Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:38.257084Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:32c1e473299639a5f851beec31438bac2d70a35f2c8f307b55c38de5fcf40d36","observation_id":"4bda4121-31bf-4c21-ba78-9b4c6e38d718","resolution":{"observed_at":"2026-08-06T21:43:38.257084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:46.762709Z","title":null,"venue":null,"work_id":"f2ae5f60-bec7-4470-8a2c-4b542db467b4","year":2014},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:38.402880Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:770c13f40cbdaf091a208ecbd57980d16631bb5463140d2d9f069d72d7fdaa04","observation_id":"483a487d-46e1-4255-b7b5-33fa0b3b4137","resolution":{"observed_at":"2026-08-06T21:43:46.876322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:38.582482Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:38.582482Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:9221d3677bb0ff7551454a26cb744e3a8c6d0e27d8f084ce4960eb81f1d10b15","observation_id":"a924e22a-9008-4b74-82f2-cb3079ae35cf","resolution":{"observed_at":"2026-08-06T21:43:38.582482Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:46.457483Z","title":null,"venue":null,"work_id":"adaf1e9b-74a9-4591-b63a-e851fa6e0aa8","year":2014},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:38.788471Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:c721488303f2ca08de87024361378f820c8e218894976dd92c30f95656929bd5","observation_id":"174fdf78-60e7-45be-a6a3-2c8189526d83","resolution":{"observed_at":"2026-08-06T21:43:46.598042Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:46.205915Z","title":null,"venue":null,"work_id":"b45e3af5-f045-423c-9751-1ce10e96458e","year":2016},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:38.908575Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:88a3431099a3dfac59c6ece5a1ea130cafab0a85984cccf87b4ae25551548be1","observation_id":"dfabe30b-ac9d-41dc-821d-f97946597e2e","resolution":{"observed_at":"2026-08-06T21:43:46.269639Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:46.016394Z","title":null,"venue":null,"work_id":"08558afd-6717-4a9f-a349-356bc1568a64","year":1999},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.021122Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:5827dd28a013926dc2ea8a6ec22eafe338d86081dec6417e8af4e34e8d751d95","observation_id":"f49592ae-1fbe-42a4-9a30-27f8ccbe6ee4","resolution":{"observed_at":"2026-08-06T21:43:46.121174Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:45.677816Z","title":"Practices for governing agentic ai systems","venue":null,"work_id":"c9e1d7a8-d7aa-48be-9c77-0b7c33a4bc10","year":null},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.155956Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:296eb83ec88f05e64e836a3f49f1c6484feed4de94264b88fc0e34ee7fbb5ec4","observation_id":"dc354d82-956e-46b0-bd62-0867290ba786","resolution":{"observed_at":"2026-08-06T21:43:45.843799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:45.501017Z","title":null,"venue":null,"work_id":"49bd4934-d262-4596-9ce1-c097c14d734d","year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.303135Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:e3d73e65b17d92ec44129205b8433a05f85f2f7d9cb75419714b68cb80c6dc80","observation_id":"9c23ad6c-3e8f-463d-a651-52d33676fa7e","resolution":{"observed_at":"2026-08-06T21:43:45.585026Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:45.182785Z","title":null,"venue":null,"work_id":"fa31bcce-be13-4f9e-84b2-c217cb8f0638","year":2010},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.468704Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:09702034373cf5bbb0bd6e390d0117cb777dc67b438df646d63cbf8124a2ff80","observation_id":"f4c85996-2a00-414a-a2b9-e99279b16178","resolution":{"observed_at":"2026-08-06T21:43:45.357904Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:44.913827Z","title":null,"venue":null,"work_id":"f032a9c6-12a5-4d1d-9432-4d1f0d32aba1","year":2018},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.557291Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:06d3d18d0d158ef6cb631b64d93ba17a302a3fb81bf24218d80c84df3cd5edf5","observation_id":"945af8fe-5db2-4067-9638-7ae1ff1bdfc6","resolution":{"observed_at":"2026-08-06T21:43:45.052750Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.13356","last_updated":"2023-10-07T09:14:43Z","snapshot_observed_at":"2026-07-06T16:22:46.711142Z","submitted_at":"2023-09-23T12:17:10Z","title":"Probing the Moral Development of Large Language Models through Defining Issues Test","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.13356","snapshot_observed_at":"2026-08-06T21:43:39.663308Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.663308Z"},"links":{"cited_paper":"/paper/2309.13356","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:ffe79c78cb28a0c66e78b9d6b9fb5bf74e9ae7b13f69aa21b89e9eb8e1a27586","observation_id":"43a11978-c6c6-4a99-a9b8-e1fef89fd303","resolution":{"observed_at":"2026-08-06T21:43:39.663308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:44.580801Z","title":null,"venue":null,"work_id":"703755e0-a815-4248-91d2-81c59a3e8758","year":2020},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.872495Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:98e188fcaafbbc8a7167a981f2f73f7a21326b2916a1592c3b8d695d8a2b5ef4","observation_id":"256934ef-e9f2-42ac-8350-ff31db532369","resolution":{"observed_at":"2026-08-06T21:43:44.713924Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:44.321046Z","title":null,"venue":null,"work_id":"192a79a8-3c58-453e-8f35-818a06234f88","year":1948},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:39.998506Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:257677f56b4d7e29a2fa5109f7f5d59b850db8f90b14dab6483ea23843bde444","observation_id":"53d6a2d8-8d93-4811-8cc3-c360f9b7d384","resolution":{"observed_at":"2026-08-06T21:43:44.471577Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:44.067147Z","title":null,"venue":null,"work_id":"8bce8b68-936c-4d99-86b6-abd6684b88cd","year":1950},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:40.093752Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:83ed6733ef3e73dfbbac1bd0290e77047908087d2a286629a4b4c00f7c941a06","observation_id":"58bd3c7b-2ad9-4907-a3d2-a3f91ab94f56","resolution":{"observed_at":"2026-08-06T21:43:44.172884Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:43.907426Z","title":null,"venue":null,"work_id":"4d790cb9-c370-43c3-8449-17e1b764961c","year":2007},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:40.265620Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:1369c717a93f11cac576847f2e0c6e8e8f9898a5f64bf9b7bfdf8fc9c1985a4d","observation_id":"69e4078e-823e-4611-b3b9-b11b2b8e7e3e","resolution":{"observed_at":"2026-08-06T21:43:44.013805Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:43.712566Z","title":null,"venue":null,"work_id":"cd46a050-1251-4468-a049-8a7b38568767","year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:40.396998Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:c6ae5d9664d7152805e58b1dfa4acf88dcb03173f802497ae8434b75bf0ad40a","observation_id":"49f5bb56-d43a-49f8-906c-bec2b54ce04d","resolution":{"observed_at":"2026-08-06T21:43:43.833508Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.04325","last_updated":"2024-06-04T22:09:46Z","snapshot_observed_at":"2026-07-06T14:15:48.224901Z","submitted_at":"2022-10-26T00:28:40Z","title":"Will we run out of data? Limits of LLM scaling based on human-generated data","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.04325","snapshot_observed_at":"2026-08-06T21:43:40.536802Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:40.536802Z"},"links":{"cited_paper":"/paper/2211.04325","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:62511b5da354b0d0640beda6f1b4554536f74b8c5b1459492a25c2d3e5f1a005","observation_id":"d0faab4c-2961-44a0-84f7-399bba3419db","resolution":{"observed_at":"2026-08-06T21:43:40.536802Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:40.701264Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:40.701264Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a83408fae4bae1b170afd9d5332b78c7c2e40f9609506d0627884b90ba6febcd","observation_id":"79fd0586-94f5-4ab4-b191-c0f5a0451bee","resolution":{"observed_at":"2026-08-06T21:43:40.701264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-07-06T06:34:26.609892Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-08-06T21:43:40.875956Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:40.875956Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:5d5b419a768d3548b0bc1b48e6f7794613d2c47185e21a64347217a6497e7b3b","observation_id":"75fba999-5051-489a-9d51-99fa4ed0ba6f","resolution":{"observed_at":"2026-08-06T21:43:40.875956Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:43.512238Z","title":null,"venue":null,"work_id":"b83093e7-0b36-43ff-952c-773215ae2cd0","year":2011},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.011460Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:aaa9195d822fd1de533ac78917a56dab01b3ea0f6a44587fee92fe2f15391fb3","observation_id":"692909f8-f2f6-44e4-873e-4b55deb4c7a7","resolution":{"observed_at":"2026-08-06T21:43:43.574933Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:43.246300Z","title":null,"venue":null,"work_id":"4f169081-1e6f-4ab5-8b9a-00b27ff253a9","year":1953},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.173567Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:ba2ad3b7a870395852d039f0dc0b8c4e2f10421a5853faba6817f4b923233956","observation_id":"a39066c8-bd37-441e-b1d0-e23805d7155a","resolution":{"observed_at":"2026-08-06T21:43:43.368445Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:41.308124Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.308124Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:352f6f44da0308a72cd1b83ec80830f413c8d800818b27a63e2c3ac3f0a22048","observation_id":"d796a52b-3899-449a-95f2-40e0b4918faa","resolution":{"observed_at":"2026-08-06T21:43:41.308124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00027","last_updated":"2025-07-13T02:13:14Z","snapshot_observed_at":"2026-07-06T19:43:09.889618Z","submitted_at":"2024-10-29T04:01:11Z","title":"Personalization of Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.00027","snapshot_observed_at":"2026-08-06T21:43:41.450167Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.450167Z"},"links":{"cited_paper":"/paper/2411.00027","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:4531ee9eb949dc234ddb5fd95e25e482a970233b08dc9ac6111df4d4653bafa3","observation_id":"b44d338d-b7d4-4e3f-807f-f87cd0a4bbbe","resolution":{"observed_at":"2026-08-06T21:43:41.450167Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:42.845784Z","title":null,"venue":null,"work_id":"ff0eb6e3-57c1-4f88-bc17-496f03d03cda","year":2024},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.596384Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:a9519bcc89b98b558e4b9c6d66d2f301b19ca306f8792fa91912ccd3e6e862ec","observation_id":"586ccf5a-1ff2-45a7-a63b-c815dfbbe265","resolution":{"observed_at":"2026-08-06T21:43:43.065499Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01964","last_updated":"2023-11-03T14:59:54Z","snapshot_observed_at":"2026-08-06T10:56:05.075839Z","submitted_at":"2023-11-03T14:59:54Z","title":"Don't Make Your LLM an Evaluation Benchmark Cheater","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01964","snapshot_observed_at":"2026-08-06T21:43:41.764235Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.764235Z"},"links":{"cited_paper":"/paper/2311.01964","citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:c98705445f85daa84c0922feb23393325cf87abc4e395c7525d6d04ce801651c","observation_id":"5ecc2e53-7f59-435d-afc4-b5bb1194e4e6","resolution":{"observed_at":"2026-08-06T21:43:41.764235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:42.603487Z","title":null,"venue":null,"work_id":"2c897ed4-cf59-4917-b6ad-64813aebcf34","year":2007},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.856323Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:70033d13aaa004ca407a932373a89ef377733780968f36fb46fcf086c1d6fe52","observation_id":"246ffb70-2d8d-4b04-824c-0f31d2261f69","resolution":{"observed_at":"2026-08-06T21:43:42.716986Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:41.969044Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:41.969044Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:4a34f7d9e9203366f0099d502ee2149785f565777fd0926377b4e29063eda486","observation_id":"289b7e1d-1326-4085-a0d8-c042bd2c8654","resolution":{"observed_at":"2026-08-06T21:43:41.969044Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:43:42.163889Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-06T21:43:42.163889Z"},"links":{"citing_paper":"/paper/2507.05266"},"observation_digest":"sha256:08bf08c8a09d498699429d57df5cb3f9d0d5e35588adf68d9f32626c8c2eebbb","observation_id":"15a5a920-5113-4418-b910-a9df9eb18613","resolution":{"observed_at":"2026-08-06T21:43:42.163889Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.05266","last_updated":"2025-06-30T06:14:32Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T21:36:14.853232Z","submitted_at":"2025-06-30T06:14:32Z","title":"User Behavior Prediction as a Generic, Robust, Scalable, and Low-Cost Evaluation Strategy for Estimating Generalization in LLMs"},"reference_resolution":{"displayed":68,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":66,"verified_exact":1,"verified_fuzzy":1},"total_outbound_references":68},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 68 of 68 outbound references and 0 inbound Pith citation observations for arXiv:2507.05266."}