{"as_of":"2026-08-22T23:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4f819d48d52b3e457e5aa64b4dc9fe3ec57723880aaede8e8d461241bd9286c1","coverage":[{"denominator":15,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-25T05:44:59.785603Z","state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.15482/citation-record","integrity":"/paper/2605.15482/integrity","json":"/paper/2605.15482/citation-record.json","paper":"/paper/2605.15482"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2109.00122","last_updated":"2022-05-07T07:52:39Z","snapshot_observed_at":"2026-08-16T17:59:38.950031Z","submitted_at":"2021-09-01T00:08:14Z","title":"FinQA: A Dataset of Numerical Reasoning over Financial Data","version":3},"cited_work":{"arxiv_id":"2109.00122","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.00122","snapshot_observed_at":"2026-07-04T04:09:34.781811Z","title":"FinQA: A dataset of numerical reasoning over financial data","venue":null,"work_id":"a11b4f4b-d693-4a54-b14a-49094201d947","year":2021},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2109.00122","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:fd1b32ba1584b7d1ed9a1c3a81ce0049e9eeafab7d85c3372e46b9bbedf71846","observation_id":"cce8f516-3b70-4efc-b8ff-86382b5402aa","resolution":{"observed_at":"2026-05-25T05:45:23.473254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03849","last_updated":"2022-10-07T23:48:50Z","snapshot_observed_at":"2026-08-20T09:14:55.571588Z","submitted_at":"2022-10-07T23:48:50Z","title":"ConvFinQA: Exploring the Chain of Numerical Reasoning in Conversational Finance Question Answering","version":1},"cited_work":{"arxiv_id":"2210.03849","doi":"10.48550/arxiv.2210.03849","metadata_source":"arxiv_reference","pith_arxiv_id":"2210.03849","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ConvFinQA: Exploring the chain of numerical reasoning in 20 conversational finance question answering","venue":"arXiv (Cornell University)","work_id":"36c74fad-5676-4ef0-bbc7-fb4f7022c0da","year":2022},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2210.03849","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:3a9e61bbc78667cd349208831ad4e86991af5fe3195f048d7e883b644361dc75","observation_id":"ca61efc2-2f1b-4055-9ca4-1b95ceefb3ee","resolution":{"observed_at":"2026-05-25T05:45:23.463886Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.07624","last_updated":"2021-06-01T05:38:50Z","snapshot_observed_at":"2026-08-16T18:24:20.042360Z","submitted_at":"2021-05-17T06:12:06Z","title":"TAT-QA: A Question Answering Benchmark on a Hybrid of Tabular and Textual Content in Finance","version":2},"cited_work":{"arxiv_id":"2105.07624","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2105.07624","snapshot_observed_at":"2026-07-01T23:06:20.514933Z","title":"TAT-QA: A question answering benchmark on a hybrid of tabular and textual content in finance","venue":null,"work_id":"1c86ee16-bad3-4e2b-9df5-bdbfa9ddb687","year":2021},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2105.07624","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:f8deb303be3ba7cd13ff78cbd0b380f307059a993dee6225fbaabce45ab342e6","observation_id":"2a9e7a2e-47d3-4455-93a1-a92bb4c0be08","resolution":{"observed_at":"2026-05-25T05:45:23.468398Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.11944","last_updated":"2023-11-20T17:28:02Z","snapshot_observed_at":"2026-08-13T18:35:52.271946Z","submitted_at":"2023-11-20T17:28:02Z","title":"FinanceBench: A New Benchmark for Financial Question Answering","version":1},"cited_work":{"arxiv_id":"2311.11944","doi":null,"metadata_source":"pith","pith_arxiv_id":"2311.11944","snapshot_observed_at":"2026-07-04T20:40:08.190653Z","title":"FinanceBench: A New Benchmark for Financial Question Answering","venue":"cs.CL","work_id":"b60d115e-50dd-42f6-9178-5d29b05e1e89","year":2023},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2311.11944","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:0d7823197604d1a05287f7c3ea2c1860f1a77e089a5e107b66b5d60413b7aef0","observation_id":"5b85d5ca-09b5-4344-a4cf-d03f9a6c67bc","resolution":{"observed_at":"2026-05-25T05:45:23.442269Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05443","last_updated":"2023-06-08T14:20:29Z","snapshot_observed_at":"2026-08-19T23:00:53.663005Z","submitted_at":"2023-06-08T14:20:29Z","title":"PIXIU: A Large Language Model, Instruction Data and Evaluation Benchmark for Finance","version":1},"cited_work":{"arxiv_id":"2306.05443","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.05443","snapshot_observed_at":"2026-07-04T15:39:57.064029Z","title":"PIXIU: A large language model, instruction data and evaluation benchmark for finance","venue":null,"work_id":"3ce38595-263a-4781-b28b-556a347b8eea","year":2023},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2306.05443","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:3098e2121f38515637d912128e82197fcd85e79c99b9513cf9a7098d075889a0","observation_id":"e45a9cb3-c7d2-4c16-a0bc-e4e795093165","resolution":{"observed_at":"2026-05-25T05:45:23.446895Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12659","last_updated":"2024-06-19T03:38:56Z","snapshot_observed_at":"2026-08-16T14:17:06.781974Z","submitted_at":"2024-02-20T02:16:16Z","title":"FinBen: A Holistic Financial Benchmark for Large Language Models","version":2},"cited_work":{"arxiv_id":"2402.12659","doi":"10.48550/arxiv.2402.12659","metadata_source":"arxiv_reference","pith_arxiv_id":"2402.12659","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"FinBen: A holistic financial benchmark for large language models","venue":"arXiv (Cornell University)","work_id":"7065aef4-c9e1-49a5-a55b-bd22a60d1389","year":2024},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2402.12659","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:b409d3fd6fb370522816ba310d4384a8831f3995127f8342f7a1d738dc9f920f","observation_id":"44fb8832-3a38-48cf-801f-86dce144c9d5","resolution":{"observed_at":"2026-05-25T05:45:23.452000Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15846","last_updated":"2025-06-18T19:54:33Z","snapshot_observed_at":"2026-08-20T18:54:07.408601Z","submitted_at":"2025-06-18T19:54:33Z","title":"Finance Language Model Evaluation (FLaME)","version":1},"cited_work":{"arxiv_id":"2506.15846","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.15846","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Finance language model evaluation (FLaME)","venue":null,"work_id":"9009b08d-b452-443e-9ffa-f5cfd916bfb5","year":2025},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2506.15846","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:875001cdf224e5c9017979d091b9dae6bc57fe2ed751af46550a39b38d46b617","observation_id":"d1d25df9-db7b-46da-aba7-53dd959a8989","resolution":{"observed_at":"2026-05-25T05:45:23.458498Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:03ed97df254954347fe024c23914b93ef050d1715f5c11cbaa644cee513838e0","observation_id":"35eff6b0-708c-42ea-a11f-a8cc05d4641f","resolution":{"observed_at":"2026-05-25T05:45:23.411757Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.11939","last_updated":"2024-10-14T18:11:58Z","snapshot_observed_at":"2026-08-21T06:07:47.921030Z","submitted_at":"2024-06-17T17:26:10Z","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","version":2},"cited_work":{"arxiv_id":"2406.11939","doi":"10.48550/arxiv.2406.11939","metadata_source":"pith","pith_arxiv_id":"2406.11939","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"From Crowdsourced Data to High-Quality Benchmarks: Arena-Hard and BenchBuilder Pipeline","venue":"cs.LG","work_id":"ad4ca175-a846-44ce-add1-5fd69a8d5c41","year":2024},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2406.11939","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:aff60cdba23806d74b411efc88820518cb9791184564a89330cab64b08edc4ae","observation_id":"23c182b7-74d4-464e-9b92-f868d5b26276","resolution":{"observed_at":"2026-05-25T05:45:23.428842Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-05-20T23:23:28.73804+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T23:23:28.73804+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05828","last_updated":"2025-08-06T17:19:50Z","snapshot_observed_at":"2026-08-17T11:07:47.822324Z","submitted_at":"2025-06-06T07:53:58Z","title":"FinanceReasoning: Benchmarking Financial Numerical Reasoning More Credible, Comprehensive and Challenging","version":2},"cited_work":{"arxiv_id":"2506.05828","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.05828","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Financereasoning: Benchmarking financial numerical reasoning more credible, comprehensive and challenging","venue":null,"work_id":"3c983517-90d3-433d-bd42-4c1afcd98c1a","year":2025},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2506.05828","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:4c0d33857508f94d0856e67409c641742e34331fa90c120bf6c29db861023e5c","observation_id":"2009fe9a-f869-4272-8385-90c97ac4791b","resolution":{"observed_at":"2026-05-25T05:45:23.433721Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.16252","doi":"10.48550/arxiv.2503.16252","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fin-R1: A large language model for financial reasoning through reinforcement learning","venue":"ArXiv.org","work_id":"81c0a7af-8b38-4e26-af7c-2042e21a23f0","year":2025},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:6d5335288d1bdb86da89af7918ada992310abb6a37a66b83e8c184c74137b559","observation_id":"023374b2-072b-4d67-8029-6a2f270f2159","resolution":{"observed_at":"2026-05-25T05:45:23.438329Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arena-hard-auto","venue":null,"work_id":"853124a1-79bd-48dc-9076-994dd954fe56","year":2026},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:c59969eb6e23eaf5f4163e00e3ca9059decff5da118f220b1f5c17d0e2a318c7","observation_id":"ee71d808-2b20-4a1d-b451-185b1616364b","resolution":{"observed_at":"2026-05-25T05:46:41.123996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:f026817aae72bac23a41ac25bca44eabc1768cda254375f978219ba257d94ec5","observation_id":"07d1c6ca-54ca-4eaa-b5e2-556521376249","resolution":{"observed_at":"2026-05-25T05:45:23.407534Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.09212","last_updated":"2024-01-17T19:09:57Z","snapshot_observed_at":"2026-08-04T18:52:19.083847Z","submitted_at":"2023-06-15T15:49:51Z","title":"CMMLU: Measuring massive multitask language understanding in Chinese","version":2},"cited_work":{"arxiv_id":"2306.09212","doi":null,"metadata_source":"pith","pith_arxiv_id":"2306.09212","snapshot_observed_at":"2026-06-29T19:43:55.030793Z","title":"CMMLU: Measuring massive multitask language understanding in Chinese","venue":"cs.CL","work_id":"30c9ec62-1af0-4f30-94b4-d1ef163eff71","year":2023},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"cited_paper":"/paper/2306.09212","citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:2ccc45ba9e48fb8c888f5f77f6a7bf0d21020c77c8e2b832383259741e7ce1cf","observation_id":"468ac5eb-bcf1-4706-8e78-5a052c94251c","resolution":{"observed_at":"2026-05-25T05:45:23.419108Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.15337","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reasoning models are test exploiters: Rethinking multiple choice","venue":null,"work_id":"b67e808d-65a2-47ec-ad66-9ce96b7ed2eb","year":2025},"citing_paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-25T05:44:59.785603Z"},"links":{"citing_paper":"/paper/2605.15482"},"observation_digest":"sha256:eaae042cf801653d2ffb27018101b00f79219d1be1d64eb4ef71f10c277f43e2","observation_id":"3ca63e0e-de68-41b3-8444-6e336d190ce0","resolution":{"observed_at":"2026-05-25T05:45:23.423999Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.15482","last_updated":"2026-05-21T18:49:22Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-19T09:54:23.087903Z","submitted_at":"2026-05-14T23:53:51Z","title":"FINESSE-Bench: A Hierarchical Benchmark Suite for Financial Domain Knowledge and Technical Analysis in Large Language Models"},"reference_resolution":{"displayed":15,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":14,"verified_fuzzy":1},"total_outbound_references":15},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 15 of 15 outbound references and 0 inbound Pith citation observations for arXiv:2605.15482."}