{"as_of":"2026-08-06T06:12:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8f3528caca0317e0f826853196bbad8de9670ce1a5fedd0c505f4805008370ba","coverage":[{"denominator":19,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":19,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T12:28:32.395213Z","state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T23:21:25.985017Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":19,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2401.02954","last_updated":"2024-01-05T18:59:13Z","snapshot_observed_at":"2026-08-02T13:11:16.882565Z","submitted_at":"2024-01-05T18:59:13Z","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-11T06:08:05.550346Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2401.02954"},"observation_digest":"sha256:21248a18457a013b585fba3b3272c09ce81a9172c7fc3700c8d90b5bdff3973a","observation_id":"42c25863-cfe0-47a6-94c5-2b7f29e04c1c","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2403.04652","last_updated":"2025-01-21T10:12:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-07T16:52:49Z","title":"Yi: Open Foundation Models by 01.AI","version":3},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-13T05:47:27.775529Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2403.04652"},"observation_digest":"sha256:f863ed7441996a0d2fb2a1d3c213b68f2ae9ce746c86b232cfe9f244cfea6f58","observation_id":"1cd82ad0-a844-4020-973b-484bf5de6693","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2405.04434","last_updated":"2024-06-19T06:04:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","version":5},"reference_index":121,"source":"arxiv_source","source_observed_at":"2026-05-11T05:36:26.207359Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2405.04434"},"observation_digest":"sha256:fbcc9a2005ccb41e3110f6db14ccb141546269695fd679a50168bccd3e9d8952","observation_id":"0ec704b1-b80c-474d-926c-be559c777325","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2411.10442","last_updated":"2025-04-07T09:09:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-15T18:59:27Z","title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","version":2},"reference_index":116,"source":"pdf_text","source_observed_at":"2026-05-16T09:16:17.150383Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2411.10442"},"observation_digest":"sha256:b6efc3d03e54749a0fabff9dc83d88905b5249edbc4ade498a140147241d5300","observation_id":"ae6bd359-f7f9-4914-8730-115862034a40","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":150,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:2e6ff4eba27984cff6d3e79bf683dacf162915eec07f910f12975b37f167a6a6","observation_id":"8613c96b-2375-4646-93c4-24db5a62c5c3","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2506.12119","last_updated":"2026-05-17T08:18:55Z","snapshot_observed_at":"2026-08-02T23:22:23.397250Z","submitted_at":"2025-06-13T17:59:05Z","title":"Mixture-of-Experts Can Surpass Dense LLMs Under Strictly Equal Resource","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-22T00:05:08.916339Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2506.12119"},"observation_digest":"sha256:80dc85953a9bcf3f0580d1770b6799e85b51ff514f494aaa885d5a4f40db93a8","observation_id":"3f5a99a0-ade0-42fc-a449-04d9131a51f9","resolution":{"observed_at":"2026-05-22T00:05:47.729661Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2507.00432","last_updated":"2025-10-20T14:27:09Z","snapshot_observed_at":"2026-07-06T21:50:16.465983Z","submitted_at":"2025-07-01T05:23:05Z","title":"Does Math Reasoning Improve General LLM Capabilities? Understanding Transferability of LLM Reasoning","version":2},"reference_index":207,"source":"arxiv_source","source_observed_at":"2026-05-19T01:01:09.840919Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2507.00432"},"observation_digest":"sha256:d1e2dcedc3d1cf000c3413f4aeec9b7487ac46a70881867176001da1f7e6af71","observation_id":"03dd5117-4e25-478c-8bb9-bfd808dda126","resolution":{"observed_at":"2026-05-19T01:01:10.027080Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T23:21:25.985017Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.05468","last_updated":"2025-08-07T15:11:17Z","snapshot_observed_at":"2026-08-05T23:21:19.135444Z","submitted_at":"2025-08-07T15:11:17Z","title":"TASE: Token Awareness and Structured Evaluation for Multilingual Language Models","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-05T23:21:25.985017Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2508.05468"},"observation_digest":"sha256:5c7e59c8dcefec72c8edec16627318381ac2329ac6c3ae6ad8da74a1c6592b8a","observation_id":"e9f3f8a1-0dd0-48bb-80cc-36c9b19869cf","resolution":{"observed_at":"2026-08-05T23:21:25.985017Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T23:04:57.082913Z","title":", author Li, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.05929","last_updated":"2025-09-10T12:09:14Z","snapshot_observed_at":"2026-08-05T23:04:47.746625Z","submitted_at":"2025-08-08T01:40:10Z","title":"Towards Reliable Generative AI-Driven Scaffolding: Reducing Hallucinations and Enhancing Quality in Self-Regulated Learning Support","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-05T23:04:57.082913Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2508.05929"},"observation_digest":"sha256:5734e77fdd05383364fde675fde542024f6a9e9653ec2546fcb1e4ff0ae9b4b6","observation_id":"310061f6-30b1-4ac3-ba9b-a8151a1c8305","resolution":{"observed_at":"2026-08-05T23:04:57.082913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T18:53:05.214761Z","title":"arXiv preprint arXiv:2305.12474","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.13938","last_updated":"2025-08-19T15:27:55Z","snapshot_observed_at":"2026-08-05T18:53:03.792366Z","submitted_at":"2025-08-19T15:27:55Z","title":"MME-SCI: A Comprehensive and Challenging Science Benchmark for Multimodal Large Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T18:53:05.214761Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2508.13938"},"observation_digest":"sha256:17da51d0a29364fc5d8f0993229ed7538a27354a3f1ae3c61e81d77f6d5847b7","observation_id":"dac95634-5e48-4fae-897d-b95e50e70bcc","resolution":{"observed_at":"2026-08-05T18:53:05.214761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":178,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:301c11f8445db0a3b12b02c399402cec0804e9280f15d9c80986a0c54d2719b0","observation_id":"ca6715dc-d740-4dac-ab4f-437a436c501e","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-04T13:15:48.547800Z","title":"Evaluating the performance of large language models on gaokao benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2510.01167","last_updated":"2026-05-30T06:40:26Z","snapshot_observed_at":"2026-08-04T13:15:39.949629Z","submitted_at":"2025-10-01T17:54:15Z","title":"Simultaneous Multi-objective Alignment Across Verifiable and Non-verifiable Rewards","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-04T13:15:48.547800Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2510.01167"},"observation_digest":"sha256:15e920a59a2a26f6fa268bbace6091dd72e179baf5cf9705d147f3ac9e5659fa","observation_id":"6ac50c29-e589-40b4-bb08-54936b5c90b9","resolution":{"observed_at":"2026-08-04T13:15:48.547800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2512.15745","last_updated":"2025-12-24T03:46:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-10T09:26:18Z","title":"LLaDA2.0: Scaling Up Diffusion Language Models to 100B","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-14T18:53:20.911374Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2512.15745"},"observation_digest":"sha256:14ae089fcc4770ed114229946c4f1039ef2ed2effad6c5363fd89518c7804145","observation_id":"ea268b62-2206-4296-8766-bdf9dc750395","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2602.00979","last_updated":"2026-05-22T03:20:51Z","snapshot_observed_at":"2026-08-03T15:56:56.036971Z","submitted_at":"2026-02-01T02:39:51Z","title":"GradingAttack: Exposing Security Vulnerabilities in LLM Based Educational Grading Agents","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-25T07:02:28.660002Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2602.00979"},"observation_digest":"sha256:13bc0224c1cb910845963aa5ef56b18c41173ac5105eeb640bdc29626aa6a012","observation_id":"b598bf54-1b21-4e92-a36c-aca177dcc8e8","resolution":{"observed_at":"2026-05-25T07:05:26.681888Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2603.18472","last_updated":"2026-04-09T02:35:56Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-19T04:08:20Z","title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","version":2},"reference_index":121,"source":"pdf_text","source_observed_at":"2026-05-15T09:11:31.870441Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2603.18472"},"observation_digest":"sha256:61ec623eb42ed65ae635be16c05f613e3798bfb6236ebed7f7bc21b943320e06","observation_id":"55e59d7a-4086-49e2-9e72-39ecb8321610","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2604.08948","last_updated":"2026-04-22T08:13:33Z","snapshot_observed_at":"2026-08-04T17:14:48.512299Z","submitted_at":"2026-04-10T04:36:21Z","title":"TaxPraBen: A Scalable Benchmark for Structured Evaluation of LLMs in Chinese Real-World Tax Practice","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-10T17:59:44.844149Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2604.08948"},"observation_digest":"sha256:c968a74c2d62405cdefa365cd12395390c609eb24fee350280c2064b854f259b","observation_id":"35fe28d5-b4bb-45b2-a6f0-ccbf62ff98e3","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2604.13515","last_updated":"2026-04-15T06:00:25Z","snapshot_observed_at":"2026-08-02T05:07:14.262340Z","submitted_at":"2026-04-15T06:00:25Z","title":"SFT-GRPO Data Overlap as a Post-Training Hyperparameter for Autoformalization","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-10T13:21:52.225115Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2604.13515"},"observation_digest":"sha256:22099832d428c1adb7796deda41b9dc1ab4275c7212a91cbcef9440bea46e10e","observation_id":"92b3a011-13f8-41b1-980d-0c82dc16e045","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2604.16392","last_updated":"2026-03-28T13:29:54Z","snapshot_observed_at":"2026-07-06T23:03:43.854612Z","submitted_at":"2026-03-28T13:29:54Z","title":"RoMathExam: A Longitudinal Dataset of Romanian Math Exams (1895-2025) with a Seven-Decade Core (1957-2025)","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-14T22:01:21.565986Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2604.16392"},"observation_digest":"sha256:9d38d4112eb924466c3f843d9d4bb15eada3d64cd2e148cf175f0ac7f9b46777","observation_id":"eabcf859-5c2b-4db3-8fa3-4864e67a1e31","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2604.20140","last_updated":"2026-04-22T03:08:30Z","snapshot_observed_at":"2026-07-06T23:06:40.154278Z","submitted_at":"2026-04-22T03:08:30Z","title":"HiPO: Hierarchical Preference Optimization for Adaptive Reasoning in LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T00:51:37.506096Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2604.20140"},"observation_digest":"sha256:a561e20a114fbd7324af972d08244096b3b4c714aef3ede228ecf21e38123575","observation_id":"5cdbe6e8-0a71-4fde-86f8-073a2554e6d1","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2604.26644","last_updated":"2026-08-03T06:41:06Z","snapshot_observed_at":"2026-08-06T05:35:06.753339Z","submitted_at":"2026-04-29T13:11:39Z","title":"When to Vote, When to Rewrite: Disagreement-Guided Strategy Routing for Test-Time Scaling","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-07T11:00:21.413246Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2604.26644"},"observation_digest":"sha256:56c914d714500475aa30213a6fc67507a93fb0a47416238c8e64d5cfc07c77d4","observation_id":"534afaf3-715d-452a-b419-7f00bad02d07","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-04T05:23:31.540079Z","title":"Zhang, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2604.26644","last_updated":"2026-08-03T06:41:06Z","snapshot_observed_at":"2026-08-06T05:35:06.753339Z","submitted_at":"2026-04-29T13:11:39Z","title":"When to Vote, When to Rewrite: Disagreement-Guided Strategy Routing for Test-Time Scaling","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T05:23:31.540079Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2604.26644"},"observation_digest":"sha256:5b3761342b7b002ed8253e6214df8a2fa9dd0ce98752c2c70d07fb80c8ac262b","observation_id":"456f76f0-19c1-4c20-8436-2b23c688c80b","resolution":{"observed_at":"2026-08-04T05:23:31.540079Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2605.04078","last_updated":"2026-05-09T13:58:28Z","snapshot_observed_at":"2026-07-06T23:16:54.673178Z","submitted_at":"2026-04-14T12:32:12Z","title":"Validity-Calibrated Reasoning Distillation","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-10T15:19:39.713044Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2605.04078"},"observation_digest":"sha256:e5151c0cb5c89e15921b8cdf3a855c778e25bcd0edd056f345cab4ca59173f70","observation_id":"70da74c3-a614-47bb-909c-1164359e15da","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2605.04078","last_updated":"2026-05-09T13:58:28Z","snapshot_observed_at":"2026-07-06T23:16:54.673178Z","submitted_at":"2026-04-14T12:32:12Z","title":"Validity-Calibrated Reasoning Distillation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-12T02:31:06.308121Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2605.04078"},"observation_digest":"sha256:db684e5efcd7e7868ca7ff74e7644c3874a945f2848aba1eb104c29f8b62e39d","observation_id":"4a4a2f6e-99c8-48e3-a917-0bfa2a78aede","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2605.08904","last_updated":"2026-05-09T11:51:34Z","snapshot_observed_at":"2026-07-06T23:21:02.177557Z","submitted_at":"2026-05-09T11:51:34Z","title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-12T02:57:15.521594Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2605.08904"},"observation_digest":"sha256:12ee527c8eb0fb1614ae4583ab6dfd3bd5529be86e11920ade5f4ef1e05a1e13","observation_id":"d8ca5f37-c0f4-41ab-8928-51e0d3ccc0aa","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2605.09635","last_updated":"2026-07-23T03:36:33Z","snapshot_observed_at":"2026-08-05T04:30:28.400226Z","submitted_at":"2026-05-10T16:24:26Z","title":"K12-KGraph: A Curriculum-Aligned Knowledge Graph for Benchmarking and Training Educational LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-12T03:46:48.498800Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2605.09635"},"observation_digest":"sha256:8a57eba8aa8c9982560910bf22fda1094d2f6c2b376c9618ff22959110f33c3e","observation_id":"0c80e584-942d-45ed-92bb-461def9eac17","resolution":{"observed_at":"2026-05-17T12:28:32.509177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-02T14:31:12.591994Z","title":"Evaluating the performance of large language models on gaokao benchmark.arXiv preprint arXiv:2305.12474, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2605.09635","last_updated":"2026-07-23T03:36:33Z","snapshot_observed_at":"2026-08-05T04:30:28.400226Z","submitted_at":"2026-05-10T16:24:26Z","title":"K12-KGraph: A Curriculum-Aligned Knowledge Graph for Benchmarking and Training Educational LLMs","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T14:31:12.591994Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2605.09635"},"observation_digest":"sha256:800271841b8f7a1056dca225a5333f640c9a182e87165ace78e5b7adff7042d5","observation_id":"5cf0c92e-db2f-428c-8fda-99a551535990","resolution":{"observed_at":"2026-08-02T14:31:12.591994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2605.26781","last_updated":"2026-05-26T09:50:35Z","snapshot_observed_at":"2026-07-06T23:36:34.652096Z","submitted_at":"2026-05-26T09:50:35Z","title":"LiveK12Bench: Have Large Multimodal Models Truly Conquered High School-level Examinations?","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T17:33:03.397468Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2605.26781"},"observation_digest":"sha256:b190921a0c6869d058d6fe873165b2734653facf1c54b623d46375e362b324c3","observation_id":"d3f92752-2110-4304-a475-22c015ef5c2e","resolution":{"observed_at":"2026-06-29T17:33:45.008370Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2606.15079","last_updated":"2026-06-13T03:21:49Z","snapshot_observed_at":"2026-07-06T23:52:28.557859Z","submitted_at":"2026-06-13T03:21:49Z","title":"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale","version":1},"reference_index":125,"source":"arxiv_source","source_observed_at":"2026-07-02T22:10:59.568675Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2606.15079"},"observation_digest":"sha256:7efdb7fa064caf8d46e5fe9d3fe9a17a3a06ac4aa6e7ebb463d67158cd9cbeae","observation_id":"1b7ea2fe-d53d-4601-a505-9f236ce6229f","resolution":{"observed_at":"2026-07-02T22:17:24.928692Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":"2305.12474","doi":"10.48550/arxiv.2305.12474","metadata_source":"pith","pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","venue":"cs.CL","work_id":"2eacfa63-8867-4a90-9bcc-52f085f33cef","year":2023},"citing_paper":{"arxiv_id":"2607.02118","last_updated":"2026-07-02T12:53:41Z","snapshot_observed_at":"2026-08-03T21:58:44.127238Z","submitted_at":"2026-07-02T12:53:41Z","title":"Enhancing Fitness Intelligence through Domain-Specific LLM Post-Training","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-07-03T13:11:44.893703Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2607.02118"},"observation_digest":"sha256:d56c3a903bf90ded3a2da9fa077e9a221d95ddfd284e622556e87f4b3745e370","observation_id":"e29a27f3-24a8-4962-837b-10c619cf0093","resolution":{"observed_at":"2026-07-03T13:18:12.265550Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-01T12:54:36.403464Z","title":"Evaluating the performance of large language models on gaokao benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.19313","last_updated":"2026-07-21T17:28:40Z","snapshot_observed_at":"2026-08-06T02:26:15.730077Z","submitted_at":"2026-07-21T17:28:40Z","title":"Off-Context GRPO: Learning to Reason on Hard Problems using Privileged Information","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-01T12:54:36.403464Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2607.19313"},"observation_digest":"sha256:023bc5b47dd60526316a18304a265337b6def783569db805e25d834ec6398299","observation_id":"9a1ecbb9-838e-406b-acd3-c244cf0f52fc","resolution":{"observed_at":"2026-08-01T12:54:36.403464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-01T06:14:29.370407Z","title":"Evaluating the performance of large language models on gaokao benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22002","last_updated":"2026-07-24T06:04:04Z","snapshot_observed_at":"2026-08-03T21:21:44.044756Z","submitted_at":"2026-07-24T06:04:04Z","title":"Learning as Reasoning Unfolds: Progressive Rollout Allocation for Efficient Reinforcement Learning","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-01T06:14:29.370407Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2607.22002"},"observation_digest":"sha256:1731a7408584b3a65b590a9c540bed187e5d91b72d440e74dad3807660d33591","observation_id":"753a4f1e-a2c1-4cdb-ae93-c89f93ad31be","resolution":{"observed_at":"2026-08-01T06:14:29.370407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-04T16:29:18.638135Z","title":"arXiv preprint arXiv:2305.12474 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02024","last_updated":"2026-08-03T10:23:42Z","snapshot_observed_at":"2026-08-06T05:38:35.138330Z","submitted_at":"2026-08-03T10:23:42Z","title":"EduZone: A Framework for Evaluating LLM Safety for K-12 Students and Teachers","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-04T16:29:18.638135Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2608.02024"},"observation_digest":"sha256:d3d53ee513ad234d97f32684ca7d1c915e27cd2d4381481a42e4e12cbe25d301","observation_id":"e5e85653-5570-4a7e-956f-0437e5891064","resolution":{"observed_at":"2026-08-04T16:29:18.638135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-05T04:16:08.481230Z","title":"arXiv preprint arXiv:2305.12474 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04010","last_updated":"2026-08-04T17:59:58Z","snapshot_observed_at":"2026-08-06T05:47:33.134015Z","submitted_at":"2026-08-04T17:59:58Z","title":"ParVL: Parallel Scaling and Expandable Compute Allocation for Multimodal LLMs","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-05T04:16:08.481230Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2608.04010"},"observation_digest":"sha256:2650e13600d77b2758e5f42b3857e498276d5bf5bec8345dc7e215d590e75a8c","observation_id":"feebda8d-256e-41e7-8181-d1053c14fbc5","resolution":{"observed_at":"2026-08-05T04:16:08.481230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2305.12474/citation-record","integrity":"/paper/2305.12474/integrity","json":"/paper/2305.12474/citation-record.json","paper":"/paper/2305.12474"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Dan Hendrycks, Collin Burns, Steven Basart, Andy Zou, Mantas Mazeika, Dawn Song, and Jacob Steinhardt","venue":null,"work_id":"7165db1f-9442-4e9a-8a0d-43246db1d50f","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:6cd88a818587ff492e9a42ff4487167353b069e11cc1dc262386fc70554ccdc8","observation_id":"5da42aed-6caf-46c4-a61f-bfa49b6d6a31","resolution":{"observed_at":"2026-05-17T12:28:32.495884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:569b899d1b5d234562b65b43a4b0e56ed354d0cb6c3c87d0de23625d48993131","observation_id":"1a8e6288-784b-4c65-83d3-4f31f40cdeb7","resolution":{"observed_at":"2026-05-17T12:28:32.420686Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04615","last_updated":"2023-06-12T17:51:15Z","snapshot_observed_at":"2026-07-06T13:19:12.109592Z","submitted_at":"2022-06-09T17:05:34Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","version":3},"cited_work":{"arxiv_id":"2206.04615","doi":"10.1162/tacl_a_00688","metadata_source":"pith","pith_arxiv_id":"2206.04615","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Beyond the Imitation Game: Quantifying and extrapolating the capabilities of language models","venue":"cs.CL","work_id":"bb63abb3-0d50-4362-b97c-b5e725b03b39","year":2022},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"cited_paper":"/paper/2206.04615","citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:2584f8b1f457c6191a50f48f0665439353837b48a3202ea059d9235906267562","observation_id":"c48315cd-a1d6-454d-9eb1-1eb430b1ae93","resolution":{"observed_at":"2026-05-17T12:28:32.430613Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d2bc52d5-0e2a-4d73-969b-b1c554bb984b","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:12096a357fa7b1e29af7c2a1a8db98f37f1fad32c04c3236d12fc36106b98b83","observation_id":"2e12e333-9264-4843-8436-8c52a50f0f68","resolution":{"observed_at":"2026-05-17T12:28:32.507446Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9092f07e-a3ec-40f2-992c-08aa61280c2b","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:ee34543080b14f6b9ca2fa6d507c22c1258e6c0547571f6240b30e545a0eace6","observation_id":"8ee23b33-4950-482b-9801-a2a255f74204","resolution":{"observed_at":"2026-05-17T12:28:32.437185Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d1bfa458-2e40-4863-98b5-03814201526b","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:f63d8e836de6d83bd65f4a8cd0a2dd1a4097ca7a30e87cf5f7207aa32aff4751","observation_id":"7f894159-50dc-4532-88d9-738449ee9e38","resolution":{"observed_at":"2026-05-17T12:28:32.441022Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In order to protect this heritage while also developing tourism activities, measures need to be taken to protect the tourism resources","venue":null,"work_id":"e018ce01-0378-4f39-9a16-a1049f5e9ca3","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:824de90c07cf228454956da45312def31c73b152529756fc39d8aab20a64f9b2","observation_id":"2dc7f7f6-d212-4254-919e-b62a1eb59f3d","resolution":{"observed_at":"2026-05-17T12:28:32.446123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"91b91059-e93d-48a9-b197-8349692597ed","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:9f3b7f4e5daae8794e6ae4935e581c07903472ee5618f50d89b380339508a423","observation_id":"fdace3f6-08ac-41dd-ad0a-e5043aa0912e","resolution":{"observed_at":"2026-05-17T12:28:32.449944Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"This will enhance the cultural literacy and environmental awareness of the tourists and reduce the damage to the terraces","venue":null,"work_id":"d16131ee-e024-4f31-9a1f-c0018736e54b","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:4dc5852a946b76166e865bd3f4a0a071a58b20bc05a8cbbe06a1873ee1b49475","observation_id":"4e33bdc9-1f4b-4351-98dc-0e868daf10c2","resolution":{"observed_at":"2026-05-17T12:28:32.453796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"1c010132-42f8-4427-9e38-25c08f9036f2","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:a8b5f0b4b39cb2870139ad14306792912b91123bd9fecf46e0c46bf65d5cc593","observation_id":"39dada78-c3c9-4183-869c-eca92b621828","resolution":{"observed_at":"2026-05-17T12:28:32.457230Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"At the same time, these facilities should be planned judiciously to avoid damage to the terraces","venue":null,"work_id":"deabb47b-4760-43ae-a60b-e8d57af21f02","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:447e39cfcb406296a6078dda28517882b3918fead7bfbac88c1f45720f2357e5","observation_id":"7cd947d7-4ba4-49b6-88d3-2d4f26dc2c62","resolution":{"observed_at":"2026-05-17T12:28:32.462517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"完善 景区规划、依法保护生态环境","venue":null,"work_id":"12c846d3-c7a9-4789-844c-1fa4357bdb51","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:fd55ca69e6d7197df843aec7386d74f68834e7059bc019b069e451fdd8fb858a","observation_id":"a3ac400a-ca29-4c75-9286-86e27fc02386","resolution":{"observed_at":"2026-05-17T12:28:32.466212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"普及旅游文化环境保 护教育，提高游客对旅游资源环境保护 的意识","venue":null,"work_id":"e53e60e3-357f-4e38-9fb6-4834b2c4ecf1","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:eaebff3cbe7f3dafdbdbab738803f3d5b4758ef3ad3408515cbc3cb847b5a557","observation_id":"d84ba65c-aabf-45ed-ae82-c06c3f386dff","resolution":{"observed_at":"2026-05-17T12:28:32.469890Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"评定该‘生 态博物馆’的环境容量，对人口数量的 容纳程度，限制客流量","venue":null,"work_id":"8be72cb4-2855-450a-aac6-be693f098a37","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:79660a53573cf2e3309371a993ea783f6eac872d3d190922b54bd75d1267ae90","observation_id":"299c4ed3-e7e0-43e1-a9a8-f4b8798faa69","resolution":{"observed_at":"2026-05-17T12:28:32.473506Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"尽可能保证新建设施与景区景观相 融合","venue":null,"work_id":"3790e628-309f-4705-be10-7e1819bde0f3","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:2aa30511f805bd40cd932eb2e5f81026e03d7d311dae76af047fc12fbc10d611","observation_id":"b92bfb5d-8eac-4c8e-86f8-4e7640a18f7c","resolution":{"observed_at":"2026-05-17T12:28:32.478599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"improve the planning of the scenic area, protect the ecological environment in accordance with the law","venue":null,"work_id":"7bbd6aa4-6c7f-408b-8383-3158f46fae40","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:d5a3fb0c367085f232b57e05f0179a4fc6a4a9c6266aeebaf8f07e30c64766a6","observation_id":"113cf6b0-eb19-4257-a3b1-60c8a18062e8","resolution":{"observed_at":"2026-05-17T12:28:32.482604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"popularize education on the protection of the tourism cultural environment, raise tourists’ awareness of the protection of tourism resources and environment","venue":null,"work_id":"5e791adf-192c-4996-9233-e84cb82be591","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:8568c557a270b25017a203f87fe7ad10ea833b9d445defeb00a5dd590a2696ca","observation_id":"2c8949de-7100-4f53-a883-3c03366751a6","resolution":{"observed_at":"2026-05-17T12:28:32.487677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"assess the environmental capacity of this ‘Ecological Museum’, regulate the carrying capacity in terms of population, limit the flow of visitors","venue":null,"work_id":"efcb772f-aa5d-4796-9dd7-79ef79fc62b2","year":null},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:13213bcab8a985c6256f18022805954f0d0790903772e16f56159c7bbb317fe0","observation_id":"920ba379-6e08-437c-acc4-4c22993a1574","resolution":{"observed_at":"2026-05-17T12:28:32.491874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ensure new facilities blend harmoniously with the scenic landscape","venue":null,"work_id":"da53808a-2e46-438a-ae16-3ad3d41cff59","year":2069},"citing_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T12:28:32.395213Z"},"links":{"citing_paper":"/paper/2305.12474"},"observation_digest":"sha256:5ff043571a73a9d53ea3150c3ef816dfc32bd1ebc3981dc434342d656c3d5772","observation_id":"ea8ba8d1-5dab-498d-99f4-bc6d9ff60ae4","resolution":{"observed_at":"2026-05-17T12:28:32.503672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-02T10:07:31.047425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark"},"reference_resolution":{"displayed":19,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":5,"verified_exact":1,"verified_fuzzy":12},"total_outbound_references":19},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 19 of 19 outbound references and 33 inbound Pith citation observations for arXiv:2305.12474."}