{"as_of":"2026-08-13T04:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c10caf126d52de0fb5eb99f5e723311117f71caad57e3debb1c22121be149d5b","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T16:30:34.819197Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2412.10056/citation-record","integrity":"/paper/2412.10056/integrity","json":"/paper/2412.10056/citation-record.json","paper":"/paper/2412.10056"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.527170Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.527170Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:46efc509560e47f4be9f04cd671b3a79d955b6c8ffc3c33f7538f043bcb4ce24","observation_id":"f9f34630-1069-48fd-a18f-d4f3e35f5d54","resolution":{"observed_at":"2026-08-11T16:30:34.527170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.534370Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.534370Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:d556d403f28a3ae0ce28fcf001dccd2f8c218dbb1c6fa253a3237e4ad802aa4d","observation_id":"1beebf1a-d41d-4f68-ac14-b04e49435939","resolution":{"observed_at":"2026-08-11T16:30:34.534370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-11T16:30:34.539948Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.539948Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:ea101cc133ecbe21f5c280d622fd2d3706e07d8c39b339552bb536d3f050fb62","observation_id":"d06d8d1f-4f0e-4276-aa29-b4a44dac2429","resolution":{"observed_at":"2026-08-11T16:30:34.539948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.921650Z","title":"a is b\" fail to learn","venue":null,"work_id":"d736be46-feb2-4ad0-9b55-64efde0d92a7","year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.545544Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:3e3ff916611aa63d3bff311edeb33507ff3bc59591b146844d7dab417262a60a","observation_id":"ebbe3dba-72e6-4765-a642-9988a64a8cb9","resolution":{"observed_at":"2026-08-11T16:30:35.927480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.902850Z","title":"Applying the rasch model: fundamental measurement in the human sciences, 2007","venue":null,"work_id":"b8fda1b6-ba60-4164-9195-fa1afa4995d1","year":2007},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.551170Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:b7fcc885a5b8b049c55936d6389514ca0379389ff24bcbc132ea3f4019fe7b8e","observation_id":"238b180c-a372-431d-81a7-b42414758e1b","resolution":{"observed_at":"2026-08-11T16:30:35.908255Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2017.14168","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.357456Z","title":"Boone and Amity Noltemeyer","venue":null,"work_id":"e3de29ab-e0df-4d6e-83ff-db4c8cd9e310","year":2017},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.556401Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:7365c454b0dcd973be2c7e5ffa32fe48de7ad84f46756aec40c8d3298b789dd7","observation_id":"6ae448e2-4f2d-43ef-b19d-74d80cf957bd","resolution":{"observed_at":"2026-08-11T16:30:35.367589Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.561380Z","title":"Internlm2 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.561380Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:92f822d2c28d68b8ab70b80830d2e52adf450f9b55cea1c651d39b7f1775409a","observation_id":"720f12fb-7884-4046-8d3d-c849f7bb9c8b","resolution":{"observed_at":"2026-08-11T16:30:34.561380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16473","last_updated":"2024-05-26T07:56:30Z","snapshot_observed_at":"2026-08-12T23:57:52.483265Z","submitted_at":"2024-05-26T07:56:30Z","title":"M$^3$CoT: A Novel Benchmark for Multi-Domain Multi-step Multi-modal Chain-of-Thought","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.16473","snapshot_observed_at":"2026-08-11T16:30:34.567451Z","title":"M ^3 cot: A novel benchmark for multi-domain multi-step multi-modal chain-of-thought, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.567451Z"},"links":{"cited_paper":"/paper/2405.16473","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:050923aa8f38d4f39216dbcab4293b9ac424d8d9d141faae32c5e78604e610bb","observation_id":"df7ad01c-87f9-45a1-9afe-1dd4a27d19dc","resolution":{"observed_at":"2026-08-11T16:30:34.567451Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08978","last_updated":"2024-10-01T01:40:14Z","snapshot_observed_at":"2026-08-12T23:02:26.524652Z","submitted_at":"2024-08-16T19:01:52Z","title":"See What LLMs Cannot Answer: A Self-Challenge Framework for Uncovering LLM Weaknesses","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08978","snapshot_observed_at":"2026-08-11T16:30:34.573319Z","title":"See what llms cannot answer: A self-challenge framework for uncovering llm weaknesses","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.573319Z"},"links":{"cited_paper":"/paper/2408.08978","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:db79d3642f69c64632a08555e66244d7185280566f7d73720f82ad2e0c702200","observation_id":"2a341b58-3540-4bf4-b95a-95630a3b55c8","resolution":{"observed_at":"2026-08-11T16:30:34.573319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.582861Z","title":"Flash A ttention-2: Faster attention with better parallelism and work partitioning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.582861Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:323a7377203c112d90ef89bbf381ae3e29f6ed7b0c03f2094635abdb13687f73","observation_id":"0680fa2f-fecc-4b3b-af2a-2c72670adfba","resolution":{"observed_at":"2026-08-11T16:30:34.582861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03438","last_updated":"2023-12-01T01:27:37Z","snapshot_observed_at":"2026-08-12T13:09:32.469432Z","submitted_at":"2023-06-06T06:35:27Z","title":"Large Language Models of Code Fail at Completing Code with Potential Bugs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03438","snapshot_observed_at":"2026-08-11T16:30:34.588170Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.588170Z"},"links":{"cited_paper":"/paper/2306.03438","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:16a0567ba825470a3a2eff67d5ff82e7e14cc659b4d1c7989d728b245a4c0672","observation_id":"3a118e8d-5923-451d-ae2d-a38037b5c9f7","resolution":{"observed_at":"2026-08-11T16:30:34.588170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-10T16:40:37.411115Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-11T16:30:34.593851Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.593851Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:2a674dbd7d63001f1848321c5e4c39d1e6e64d1f2b4ae374477e9073b8d8a6d9","observation_id":"812d37ec-fa00-4e6a-8a5a-ed9a8814733b","resolution":{"observed_at":"2026-08-11T16:30:34.593851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.599529Z","title":"Chatglm: A family of large language models from glm-130b to glm-4 all tools, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.599529Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:9971d38c2bbbcbc419addfb03cd7dd6a4c4367a34c771275d8d7bafc9d45e68f","observation_id":"3243e408-c118-4e2f-a9e5-379ce6eab065","resolution":{"observed_at":"2026-08-11T16:30:34.599529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.604432Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.604432Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:8e01b8163ae71a5d9d213b9edfcc8956b7e455b4ab1f371d2a2bedee033d0eb8","observation_id":"ea5a6b41-6cc2-4f67-9e9c-6d29285933cb","resolution":{"observed_at":"2026-08-11T16:30:34.604432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.834735Z","title":"Measuring massive multitask language understanding, January 2021 b","venue":null,"work_id":"f42f7e92-5434-4b65-b079-b0239de52afe","year":2021},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.611093Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:0b6b94e9107df49fdb7024711fc28100f9fe1043cbf42f50810429e3becaf42d","observation_id":"d8de0acd-576a-40d2-846b-b7107dce6c1e","resolution":{"observed_at":"2026-08-11T16:30:35.840345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.615932Z","title":"Measuring mathematical problem solving with the math dataset","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.615932Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:ce2dd2cea52d18f8d3ca3c39ff71b8938fcf16e72086f2a37337ef6a7d37f20c","observation_id":"f9c0d431-1110-40f5-a877-8be71f50a559","resolution":{"observed_at":"2026-08-11T16:30:34.615932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.800389Z","title":"C-eval: A multi-level multi-discipline chinese evaluation suite for foundation models","venue":null,"work_id":"a878cea3-8a3f-4dfc-a472-64d1081486e2","year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.620810Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:ff9ce8753b9583eedd250322a1f654e13c361461d1ebc9279b1ef2162969202e","observation_id":"bb6e4b87-812c-471b-83af-6a2bf5457a75","resolution":{"observed_at":"2026-08-11T16:30:35.805435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.781199Z","title":"New Ontology and Knowledge Graph for University Curriculum Recommendation","venue":null,"work_id":"d4f46e83-b75f-45b4-92cd-081bc4a2a5b2","year":2022},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.627029Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:08ce6559509980af8f6ca4d7f06ff12d0ebdbd71382aad0e7e8d84ab2f18dbae","observation_id":"b2bea7fb-0b39-46d5-a9a7-5182f0f61443","resolution":{"observed_at":"2026-08-11T16:30:35.787252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.762615Z","title":"Large language models and simple, stupid bugs","venue":null,"work_id":"d2b44208-576c-42af-8a3f-ea239c8235e9","year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.631971Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:967bf4824b09fb68213204e8eec565fd0d684c01f6a428a6d1fd225e1997d8fc","observation_id":"699aae21-23b2-4c57-9c14-495595825d32","resolution":{"observed_at":"2026-08-11T16:30:35.768425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.745486Z","title":"FigureQA : An annotated figure dataset for visual reasoning, February 2018","venue":null,"work_id":"3b265ba0-cba1-4ce3-a307-ae8f8753beb4","year":2018},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.636984Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:1ae4d6a0b897defdd2f24c881c4dc7162c73db2ef0ef5197868aff65a0a6a491","observation_id":"ab75a756-39e4-42ed-b87e-7a1b909235b4","resolution":{"observed_at":"2026-08-11T16:30:35.750702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.726099Z","title":"A diagram is worth a dozen images, March 2016","venue":null,"work_id":"5a445790-afa7-4f32-b207-094bc2181b16","year":2016},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.642021Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:888e88bbc078712ca2b9988f11ee2f632669660ef377d0f6f56f8c1536f18938","observation_id":"acbf3732-4f48-4231-acbd-b8cf6b609d78","resolution":{"observed_at":"2026-08-11T16:30:35.731918Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-981-15-1800-3","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.916806Z","title":"Rasch Measurement: Applications in Quantitative Educational Research, volume 1 of Education","venue":null,"work_id":"ea8b8525-a2b6-4768-b724-3afb0a536e5b","year":2020},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.646882Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:71f3785b46a03b08f7ca6b8d71ae62b4c8c890aa247f0bc57d39286305630fe5","observation_id":"50d3fac8-5555-4968-b1b9-419893fedc70","resolution":{"observed_at":"2026-08-11T16:30:34.924315Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.652187Z","title":"Cmmlu: Measuring massive multitask language understanding in chinese, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.652187Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:c563bd6dcf181e182f2cb9859684f1a750d325430b532c52585af4eb43a4a0e3","observation_id":"c2524c55-333c-49c2-81fa-21fcdde0b44c","resolution":{"observed_at":"2026-08-11T16:30:34.652187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.694328Z","title":"CMMLU: measuring massive multitask language understanding in chinese","venue":null,"work_id":"384d5ab8-df15-43f8-9d50-a4827a4030b5","year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.657090Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:8eb42fd0d36f4c7f32e50098c8b9b87d97c84f14d083ac9a77b801f1fc0223ad","observation_id":"56a7f43f-c67d-4643-9704-7c7807fcb0d4","resolution":{"observed_at":"2026-08-11T16:30:35.700355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.661779Z","title":"Truthfulqa: Measuring how models mimic human falsehoods","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.661779Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:db4288629355f15a358bc5e33c8c424031829ec5ec0f13ecb997195f2fe1cdc8","observation_id":"862c2ae7-19d3-41f8-8a04-45789ea16988","resolution":{"observed_at":"2026-08-11T16:30:34.661779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.677047Z","title":"MMBench : Is your multi-modal model an all-around player?, August 2024","venue":null,"work_id":"94d38a3e-1562-4d15-af0f-dbe29e7c3a50","year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.666335Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:2c3b8df30d0589f96fb406d4aed77bf3c24ae5a1fa2ee18e2aee7f6ac43555aa","observation_id":"e502e362-2694-42b3-b138-58a17e22547e","resolution":{"observed_at":"2026-08-11T16:30:35.682459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.659382Z","title":"Learn to explain: Multimodal reasoning via thought chains for science question answering, October 2022 a","venue":null,"work_id":"a83e5c94-a963-4adc-8400-3c3ea203ce9b","year":2022},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.670919Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:d8511da9cefc746b3a8397a91340ca57f62971e83d82f430a86d3a70d6f87a32","observation_id":"ab262c57-8b40-40e5-b69e-1991f1982407","resolution":{"observed_at":"2026-08-11T16:30:35.664577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.642390Z","title":"IconQA : A new benchmark for abstract diagram understanding and visual language reasoning, July 2022 b","venue":null,"work_id":"b939e0df-78bf-4ff4-bd5f-64082d04e668","year":2022},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.675623Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:e20003c5805f038d8f4a46084f72064b6a61d02596a38c98c92179fa8883a89a","observation_id":"3edc29f7-0156-48a2-8c94-6fb23322fa05","resolution":{"observed_at":"2026-08-11T16:30:35.647385Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.680434Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.680434Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:67ab2e904ad045bdb29af39b20add34b842f3ce39ca3027fb267256e9a1e6cb4","observation_id":"1f8d3dba-57da-4986-99c8-fe07a68f3bb9","resolution":{"observed_at":"2026-08-11T16:30:34.680434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.607939Z","title":"OK-VQA : A visual question answering benchmark requiring external knowledge, September 2019","venue":null,"work_id":"3132cc70-18b0-4f9e-9903-ffd1dc8b364e","year":2019},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.685200Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:7505203258189326b0af5096bee67970604ffa5b6174a16d2fe549cf62e65081","observation_id":"8c248b91-8ab9-4a96-a7f0-663ef60f835a","resolution":{"observed_at":"2026-08-11T16:30:35.615582Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.588200Z","title":"Mistral Large 2: Designed for Single-Node Inference with Long-Context","venue":null,"work_id":"ae513dd0-1f69-4061-8580-b21fe80928ec","year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.690073Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:9289dda63a3cdae1e317167807f0b4ffa2dd79144f9925cc89c8b76932b57a8b","observation_id":"b4c773d5-cbb0-400c-8c23-35915940be86","resolution":{"observed_at":"2026-08-11T16:30:35.594683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01790","last_updated":"2025-02-28T02:40:58Z","snapshot_observed_at":"2026-08-12T22:52:27.327127Z","submitted_at":"2024-09-03T11:09:44Z","title":"Training on the Benchmark Is Not All You Need","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01790","snapshot_observed_at":"2026-08-11T16:30:34.695025Z","title":"Training on the benchmark is not all you need","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.695025Z"},"links":{"cited_paper":"/paper/2409.01790","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:36c0d644d731462b1068b87ae23b4f22e6c171517bc42b7c2cdc920eaf6fd903","observation_id":"17a8cc32-8cd0-499f-9d89-3ab18a11bf20","resolution":{"observed_at":"2026-08-11T16:30:34.695025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T16:30:34.702726Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.702726Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:70ee8e5d0cf86227c13aa35ff2c30b030c8304237c73fd63bd4f8cb30677a914","observation_id":"dc5ec2c7-f3f5-437f-b7e4-f7fd51ea6349","resolution":{"observed_at":"2026-08-11T16:30:34.702726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01284","last_updated":"2024-07-01T13:39:08Z","snapshot_observed_at":"2026-08-12T06:11:26.786967Z","submitted_at":"2024-07-01T13:39:08Z","title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01284","snapshot_observed_at":"2026-08-11T16:30:34.708594Z","title":"We-math: Does your large multimodal model achieve human-like mathematical reasoning? arXiv preprint arXiv:2407.01284, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.708594Z"},"links":{"cited_paper":"/paper/2407.01284","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:1cace9ffa76e7032ece14304981d0dbb5639634decb48497c3a7950a93236eee","observation_id":"cfc35231-87ab-431a-b617-2423d282763e","resolution":{"observed_at":"2026-08-11T16:30:34.708594Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.713913Z","title":"Probabilistic models for some intelligence and attainment tests","venue":null,"work_id":null,"year":1993},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.713913Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:18fd27e6641b743153a69829c80a08c97a6c5d890dae6859723d11ce6880e79a","observation_id":"586c737e-b5d0-44fa-b697-59adf5aaeac8","resolution":{"observed_at":"2026-08-11T16:30:34.713913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.718989Z","title":"Winogrande: An adversarial winograd schema challenge at scale","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.718989Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:2f17e60030d29c68f19951ec9316be1ae5f23ef3cc8e220331fc8f9a5f2f0a78","observation_id":"84fef0db-e386-495d-86d6-3aad81f94a37","resolution":{"observed_at":"2026-08-11T16:30:34.718989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.17294","last_updated":"2024-10-08T06:58:27Z","snapshot_observed_at":"2026-08-12T23:35:33.409232Z","submitted_at":"2024-06-25T05:43:21Z","title":"Math-LLaVA: Bootstrapping Mathematical Reasoning for Multimodal Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.17294","snapshot_observed_at":"2026-08-11T16:30:34.724510Z","title":"Math-llava: Bootstrapping mathematical reasoning for multimodal large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.724510Z"},"links":{"cited_paper":"/paper/2406.17294","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:178ab3f1573f40e7ecd844aee8f5667a8ce1bc58fa12801967de5edfefaf6843","observation_id":"41d8d7db-ae83-4c73-b371-b63728d92ebb","resolution":{"observed_at":"2026-08-11T16:30:34.724510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.729872Z","title":"Assessing programming task difficulty for efficient evaluation of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.729872Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:a2996d2bdd7dbf424f417389f9c28ddfd91c37f1e149dbcd6d97d07687309e1a","observation_id":"ef9a2e91-6618-47de-a8f3-eadb8d6e1dd0","resolution":{"observed_at":"2026-08-11T16:30:34.729872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.734760Z","title":"Internlm: A multilingual language model with progressively enhanced capabilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.734760Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:9e7d8f12d086dfaa134f60870d7d96b0414420907d8f549038339ad82368b76a","observation_id":"eadb0a71-d0e0-4b1e-922a-da1e1c8867d3","resolution":{"observed_at":"2026-08-11T16:30:34.734760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.739856Z","title":"Mmlu-pro: A more robust and challenging multi-task language understanding benchmark, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.739856Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:1b5073dbfb373e917d6093b06d582d1a6286c9b946db583bdc86ea115a7c0645","observation_id":"382f101f-6b49-47e6-8b35-2688d437a679","resolution":{"observed_at":"2026-08-11T16:30:34.739856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.524701Z","title":"Reasoning or reciting? exploring the capabilities and limitations of language models through counterfactual tasks","venue":null,"work_id":"4298f349-ddf7-4160-903a-5f64b85832e7","year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.745082Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:e0a96819fe234a75b604bd8fb5753952d670762cd7630d962895c92e177b0361","observation_id":"cddee14a-bbc2-48d8-b72c-7f28dd03e332","resolution":{"observed_at":"2026-08-11T16:30:35.530830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-11T16:30:34.752029Z","title":"Qwen2 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.752029Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:304c367260de46986380a2d9bc3acef9b8c21f127e5c15891f23110397a5f01b","observation_id":"981a2628-7dfc-43d5-8c5c-7073ee72c713","resolution":{"observed_at":"2026-08-11T16:30:34.752029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.757147Z","title":"Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.757147Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:0858a0c39498d29ece3156004283ca1aae1970bbcc518415c694a17878428c7d","observation_id":"a546928d-c373-40bd-9018-3262832ed61d","resolution":{"observed_at":"2026-08-11T16:30:34.757147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.762725Z","title":"Hellaswag: Can a machine really finish your sentence? In Anna Korhonen, David R","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.762725Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:9e81c83b04819d785d29b0f6c72f515b41ce22e88c112e61759c277cf7d040bc","observation_id":"b024bb3d-431b-4fd1-996f-03166bc670d0","resolution":{"observed_at":"2026-08-11T16:30:34.762725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.12474","last_updated":"2024-02-24T15:44:21Z","snapshot_observed_at":"2026-08-08T08:58:54.609425Z","submitted_at":"2023-05-21T14:39:28Z","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.12474","snapshot_observed_at":"2026-08-11T16:30:34.768686Z","title":"Evaluating the performance of large language models on gaokao benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.768686Z"},"links":{"cited_paper":"/paper/2305.12474","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:1f0c5096fe99847f59da92f7c433fb314b3b8ad83ff29c6578ed7e62b4647cb3","observation_id":"66afb50f-3d1f-46ca-9669-24b7d6baeaa3","resolution":{"observed_at":"2026-08-11T16:30:34.768686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.494854Z","title":"Can llm replace stack overflow? a study on robustness and reliability of large language model code generation","venue":null,"work_id":"aeb4ef9c-97ff-4143-a50a-99b44df51fdf","year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.774577Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:67eff31e4595a8b629989c853454ee93742dcde14fc305624aa4e7b27171a906","observation_id":"327bc6d7-4af6-4a84-b3e3-d2b9490da6e7","resolution":{"observed_at":"2026-08-11T16:30:35.500400Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01964","last_updated":"2023-11-03T14:59:54Z","snapshot_observed_at":"2026-08-06T10:56:05.075839Z","submitted_at":"2023-11-03T14:59:54Z","title":"Don't Make Your LLM an Evaluation Benchmark Cheater","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01964","snapshot_observed_at":"2026-08-11T16:30:34.780367Z","title":"Don't make your llm an evaluation benchmark cheater","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.780367Z"},"links":{"cited_paper":"/paper/2311.01964","citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:e89babc767dfefeda3c7c0cec17a6bd4bfe52eed14cbe8146aaf4727a364b117","observation_id":"e9f9fb6f-e8ce-4b77-8a6a-5e870ea1c6ee","resolution":{"observed_at":"2026-08-11T16:30:34.780367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.786183Z","title":"Larger and more instructable language models become less reliable","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.786183Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:a4de0962a4b220b9a0b7aeeecd098915b6931c547f34468387571bc684a8af7a","observation_id":"a8208fe8-0b2f-4917-b2fa-3a15216799d2","resolution":{"observed_at":"2026-08-11T16:30:34.786183Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.474769Z","title":"Gaokao-mm: A chinese human-level benchmark for multimodal models evaluation, 2024","venue":null,"work_id":"5626335b-8913-4209-a591-76545929eddd","year":2024},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.791177Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:4176fc4f1c0dc21039b09fc5c1b991e0f8abffa067cc00a410abc5a0c2f98706","observation_id":"03664885-0d05-47c9-8ada-3b3656cf283b","resolution":{"observed_at":"2026-08-11T16:30:35.481241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.796305Z","title":"@esa (Ref","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.796305Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:9cdaf86ca5c3220c2da62cf3a630052089438ead0797c0fd9527e2238beab9ce","observation_id":"ec80e9f8-c573-412e-893c-6516311357aa","resolution":{"observed_at":"2026-08-11T16:30:34.796305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.802740Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.802740Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:adecdbe639ae80973e429683b88ab3ca53bf49b32b248dda3fc2ba699ef87fdb","observation_id":"5b9a984e-8ad1-4bc1-ab2a-e77f2536960a","resolution":{"observed_at":"2026-08-11T16:30:34.802740Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:35.431953Z","title":null,"venue":null,"work_id":"d002f06b-2396-43a6-9a50-67a3eaa64dc5","year":null},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.807826Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:3cfe3de16084435ef152455424dbf31f9be4b06bbd606a05a38dde12ce72929c","observation_id":"7a33b7fe-d241-4b37-ac08-1263e9b29ec0","resolution":{"observed_at":"2026-08-11T16:30:35.437386Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.814439Z","title":", \" * write output.state after.block = add.period write","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.814439Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:89d528c7b00102c5be2695f9d38e4a3542a47e4affadd5309a0f0d969135643a","observation_id":"f213a072-b012-4ecd-8f93-9642034b8ab2","resolution":{"observed_at":"2026-08-11T16:30:34.814439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T16:30:34.819197Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-11T16:30:34.819197Z"},"links":{"citing_paper":"/paper/2412.10056"},"observation_digest":"sha256:249fa6a13e630d09d435a13bb9c548692616d4170b117d8a95ded9e07f71567d","observation_id":"6af2e208-bf5f-4412-88c1-cf0aa85a9c7e","resolution":{"observed_at":"2026-08-11T16:30:34.819197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2412.10056","last_updated":"2024-12-13T11:38:10Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-11T21:38:17.329379Z","submitted_at":"2024-12-13T11:38:10Z","title":"GAOKAO-Eval: Does high scores truly reflect strong capabilities in LLMs?"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":35,"verified_exact":1,"verified_fuzzy":17},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 0 inbound Pith citation observations for arXiv:2412.10056."}