{"as_of":"2026-08-21T23:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:69e161a803471cc84763e68a958afb228705d1082dbc35d91b9380c0e102009c","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T21:00:08.307197Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.01783/citation-record","integrity":"/paper/2608.01783/integrity","json":"/paper/2608.01783/citation-record.json","paper":"/paper/2608.01783"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.763444Z","title":null,"venue":null,"work_id":"a329b0b1-8718-45a8-84a1-abdf3d7330c8","year":2022},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.145354Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:01638b9bca401d5c034d517b92bc4074062f8a3e2de79f78ddfe577d7fbe58bb","observation_id":"30a951e5-6d4a-4293-bf49-68b2e629d3d2","resolution":{"observed_at":"2026-08-04T21:00:08.767754Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.749072Z","title":"B., Mann, B., Ryder, N., Subbiah, M., Kaplan, J., Dhariwal, P., et al","venue":null,"work_id":"9ef5b8c2-6125-4d34-bb83-a6267393e134","year":2020},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.152708Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:7727f8162378cc97c5b2d617b9fdcaf2e0e3880ed67a67a496c4f0e3fa45a49c","observation_id":"9a5a0664-d21b-44ef-9329-1b424b12a216","resolution":{"observed_at":"2026-08-04T21:00:08.754274Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.159654Z","title":null,"venue":null,"work_id":null,"year":1968},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.159654Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:5f68085bbd40c4e816ab19badc8a4aeef2a067ef0f0ee49dcf5fb9bd7a8eb1b9","observation_id":"85972e56-26b2-46f2-81e8-2110282c0a00","resolution":{"observed_at":"2026-08-04T21:00:08.159654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.733929Z","title":null,"venue":null,"work_id":"91f360fd-e391-4d8d-8b60-8507b681a01f","year":1994},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.165183Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:c1520c6ffa34ce54fcc5c6d4bd8865b4d0141922b81d15b9ff63a71f77ab8631","observation_id":"b58fb026-5ca5-4169-8b08-80961d7b367c","resolution":{"observed_at":"2026-08-04T21:00:08.738275Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.719076Z","title":null,"venue":null,"work_id":"c486d3a8-ffd1-45b0-981a-50f8d247261c","year":2006},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.170060Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:9adea852485e38201fe792f47b6fd84bdf1df3033d75dddb7929a3afa3264516","observation_id":"a6761e1a-1069-45a4-92e5-4cd9c2abc676","resolution":{"observed_at":"2026-08-04T21:00:08.723473Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.703378Z","title":null,"venue":null,"work_id":"cc15b593-2b27-4e94-8b98-86a7fb849c5e","year":2017},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.175575Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:11b60a617d597a7b21a467c2f8c62f5232a31e88f87604afe5179420ca801c6e","observation_id":"56587aba-75b4-4f84-a03c-5892df6f90c7","resolution":{"observed_at":"2026-08-04T21:00:08.708001Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.02442","last_updated":"2025-08-04T14:02:12Z","snapshot_observed_at":"2026-08-15T07:15:56.322982Z","submitted_at":"2025-08-04T14:02:12Z","title":"Assessing the Reliability and Validity of Large Language Models for Automated Assessment of Student Essays in Higher Education","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.02442","snapshot_observed_at":"2026-08-04T21:00:08.185248Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.185248Z"},"links":{"cited_paper":"/paper/2508.02442","citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:9e988cbb4482c6e6d0a04b255d04ad4b4212e346ad1e1b4ab2e5050017138b38","observation_id":"cc89e189-7f0e-4dfe-958a-090eee4a3e8d","resolution":{"observed_at":"2026-08-04T21:00:08.185248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.687070Z","title":null,"venue":null,"work_id":"53b972fb-9a92-4888-b2a8-617fc4ddcf48","year":2025},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.190819Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:c5433591cf93ad5c590fc9ea4eeead412defff084245f877cc44190af544a6fd","observation_id":"ab9a605f-a57f-4766-a732-15beeb144c5f","resolution":{"observed_at":"2026-08-04T21:00:08.691773Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.670652Z","title":null,"venue":null,"work_id":"433eceb8-9d29-4889-815b-de16b32f6015","year":2021},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.197408Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:5efec24208976a9f76823039730402cc0181614bfcf360682d43dcac4de3d6e4","observation_id":"006af2df-823f-4c7b-9ecb-d96a431617b3","resolution":{"observed_at":"2026-08-04T21:00:08.675607Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.05553","last_updated":"2023-07-09T11:04:13Z","snapshot_observed_at":"2026-08-18T10:27:07.980184Z","submitted_at":"2023-07-09T11:04:13Z","title":"Review of feedback in Automated Essay Scoring","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.05553","snapshot_observed_at":"2026-08-04T21:00:08.207394Z","title":"J., Kim, Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.207394Z"},"links":{"cited_paper":"/paper/2307.05553","citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:a0d8d05dc3f7ded6d569334dc67ef203fdad8b7cf2ca8ab86bd0d872bc1c553e","observation_id":"efd332fc-9e92-43b2-bcfc-a979ca6054ea","resolution":{"observed_at":"2026-08-04T21:00:08.207394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.655179Z","title":"K., & Li, M","venue":null,"work_id":"8eb9b043-1966-49aa-84dd-d57884969fa5","year":2016},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.215812Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:02d806996854fe7ab9d41b3d0b4e83780edd5e864e65eeb6ba5c407c7da06b6a","observation_id":"bf7e6273-1472-4faf-9bb9-db91c77fb61e","resolution":{"observed_at":"2026-08-04T21:00:08.660668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.635892Z","title":"M., Payne, D., & Alm \\'e n, B","venue":null,"work_id":"3230c770-b989-4ea0-bb37-55110f18282f","year":2018},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.227232Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:5a592bfcb9e6e37c3eaf98ad84ac08e7271d0b3c78d9ecd2d971f67b174dbf4e","observation_id":"d52d8b1a-ee73-404f-b5a6-a40932fd6c78","resolution":{"observed_at":"2026-08-04T21:00:08.640431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.619801Z","title":null,"venue":null,"work_id":"f01fc878-6b83-46db-8e7e-a41932cbd441","year":2020},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.232292Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:65475ef4e187d45768851979e3c2c09f538dfbbd98de8bad7e3e865f89d7c22e","observation_id":"fbd9a21e-11ba-43b2-8fc7-afd7a979abd4","resolution":{"observed_at":"2026-08-04T21:00:08.624228Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.600349Z","title":null,"venue":null,"work_id":"f373741d-50fe-4457-9b79-4df38e1507b0","year":2025},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.242259Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:6df09010e26dad8371169768f490d3355901f84fcedda108bf009f3bd155dea4","observation_id":"3ec5c8b0-9052-4604-82ba-9edf27ff59ff","resolution":{"observed_at":"2026-08-04T21:00:08.605623Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.585018Z","title":null,"venue":null,"work_id":"1440de1e-f9d4-4ac2-926a-ec6698bd9dd4","year":2023},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.247602Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:21ff150a5e54bcd03ee9c4a31fa61dd4c61f1598bea71c8a24fc23d692cbe59e","observation_id":"918132a6-07c1-4aa6-ba31-33639e1aceae","resolution":{"observed_at":"2026-08-04T21:00:08.589426Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.568562Z","title":null,"venue":null,"work_id":"21d57283-9639-49d9-8e92-eae597b41840","year":2025},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.252960Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:ba4123848bb0286cd73d61f2e676dc559b535611933efc93b36849854a3538bb","observation_id":"58897323-001f-46a3-a590-7408bb4d7ea6","resolution":{"observed_at":"2026-08-04T21:00:08.573127Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.550781Z","title":"J., Correnti, R., Wang, E., & Matsumura, L","venue":null,"work_id":"28d0f379-d24e-414f-aa46-a78d1d1c403e","year":2017},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.257237Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:8deb49756b52e2c38fc70c7b9017d0f6c56603bcfb3f28c0d6c943728d343079","observation_id":"7e0c3db6-4297-4df0-bb19-a7a490038df0","resolution":{"observed_at":"2026-08-04T21:00:08.555831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.531789Z","title":null,"venue":null,"work_id":"6587d481-0344-4548-b01d-f2334b6c698a","year":2013},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.261743Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:eaa93b004694f8179d23908003df6d362e3d122c7539f076003b82c1b6e49d11","observation_id":"c836209e-d899-4a50-a491-5fb8cde55c10","resolution":{"observed_at":"2026-08-04T21:00:08.536089Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.517056Z","title":null,"venue":null,"work_id":"e7366b61-4973-4dfe-9e22-343fb7793790","year":2024},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.273684Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:4f32a37397467126172819b6d6a73eac93b261042846b590009174db3f1ac1b5","observation_id":"1cde1584-576f-4d9b-9581-fb94003409fc","resolution":{"observed_at":"2026-08-04T21:00:08.521671Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.12509","last_updated":"2025-02-18T15:24:25Z","snapshot_observed_at":"2026-08-20T17:50:08.381556Z","submitted_at":"2024-12-17T03:37:31Z","title":"Can You Trust LLM Judgments? Reliability of LLM-as-a-Judge","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.12509","snapshot_observed_at":"2026-08-04T21:00:08.279354Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.279354Z"},"links":{"cited_paper":"/paper/2412.12509","citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:00a37f1cbbf7c93edff5fedd6953f8b2da8aea7dc3c4443630016414b667ef35","observation_id":"ff66f522-10ee-44a2-83a0-0e9740e01355","resolution":{"observed_at":"2026-08-04T21:00:08.279354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.284049Z","title":null,"venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.284049Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:5196bf28e0ce2782697ffcc6a8304f8834cb55fec1289e474df92c64c7292f78","observation_id":"8aa18379-3d31-487b-ac4a-39bef4ca9cae","resolution":{"observed_at":"2026-08-04T21:00:08.284049Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.11171","last_updated":"2023-03-07T17:57:37Z","snapshot_observed_at":"2026-08-16T01:49:22.176843Z","submitted_at":"2022-03-21T17:48:52Z","title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.11171","snapshot_observed_at":"2026-08-04T21:00:08.289804Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.289804Z"},"links":{"cited_paper":"/paper/2203.11171","citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:962ad1f5610306f58664835d45e267fac51aff73362a6fc2b14249c2d6551c0b","observation_id":"fb26a0b8-56d6-4858-aed6-df6b94024721","resolution":{"observed_at":"2026-08-04T21:00:08.289804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.500511Z","title":null,"venue":null,"work_id":"e3331c3c-695c-4950-bc7a-53c7b463185f","year":2022},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.295385Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:7955a60a65eb500be230cd7fa8fe1cdf604906e6bb4a7f575ef97f38bcc93971","observation_id":"9681bbc1-9127-49ed-8c0b-d5f691341541","resolution":{"observed_at":"2026-08-04T21:00:08.505700Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.483577Z","title":"M., Xi, X., & Breyer, F","venue":null,"work_id":"28f90025-f120-4e42-8143-381c60e7a4d1","year":2012},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.301327Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:ad62c11e632778bbd78e217e87d5f0f4fdaf41fe8841301282080f5efec95f25","observation_id":"4b4e307c-8c12-4266-94c0-6e4d40de3a42","resolution":{"observed_at":"2026-08-04T21:00:08.488697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T21:00:08.456673Z","title":null,"venue":null,"work_id":"a70a041a-8908-475b-820f-a8088fc4d645","year":2025},"citing_paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-04T21:00:08.307197Z"},"links":{"citing_paper":"/paper/2608.01783"},"observation_digest":"sha256:0fe089d9124c5d29a61cd554d1613b7d15b1712b9939bab7633848aedee46b44","observation_id":"d32c9adc-e960-4bda-8bcc-e1977e71478b","resolution":{"observed_at":"2026-08-04T21:00:08.466355Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.01783","last_updated":"2026-08-03T06:58:51Z","latest_version":1,"primary_category":"cs.SD","snapshot_observed_at":"2026-08-18T01:43:48.089550Z","submitted_at":"2026-08-03T06:58:51Z","title":"Comparative Validation of GPT-4o-mini and Teacher Mean Scores for Automated Scoring of Music Analysis Responses: Single-Pass Deployment, Repeatability, and Strategy-Specific Bias"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":20,"verified_exact":0,"verified_fuzzy":5},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2608.01783."}