{"as_of":"2026-08-09T05:05:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9b9e5d28374242fe6941237cafbe16d10ee22fa408047d027464461ef6d4f53a","coverage":[{"denominator":22,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:11:46.806710Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":4,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":4,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T08:46:14.378320Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T23:29:02.503467Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"cited_work":{"arxiv_id":"2505.16113","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16113","snapshot_observed_at":"2026-07-03T23:29:02.503467Z","title":"Tools in the loop: Quantifying uncertainty of llm question answering systems that use tools","venue":null,"work_id":"70b17007-1b2c-4c43-b544-5ea8a6297c97","year":2025},"citing_paper":{"arxiv_id":"2604.23505","last_updated":"2026-04-26T02:48:03Z","snapshot_observed_at":"2026-07-06T23:09:43.659053Z","submitted_at":"2026-04-26T02:48:03Z","title":"Uncertainty Propagation in LLM-Based Systems","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-08T06:10:07.638483Z"},"links":{"cited_paper":"/paper/2505.16113","citing_paper":"/paper/2604.23505"},"observation_digest":"sha256:71436bc4bd959cce8aa1117e8547cbada719f46dce791bea0943e1954cd6e535","observation_id":"bd986cb2-9269-48f6-b838-cb2b1614f789","resolution":{"observed_at":"2026-05-11T21:16:23.687803Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"cited_work":{"arxiv_id":"2505.16113","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16113","snapshot_observed_at":"2026-07-03T23:29:02.503467Z","title":"Tools in the loop: Quantifying uncertainty of llm question answering systems that use tools","venue":null,"work_id":"70b17007-1b2c-4c43-b544-5ea8a6297c97","year":2025},"citing_paper":{"arxiv_id":"2606.18467","last_updated":"2026-06-16T20:27:37Z","snapshot_observed_at":"2026-08-03T19:18:31.787087Z","submitted_at":"2026-06-16T20:27:37Z","title":"ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-26T22:16:17.385766Z"},"links":{"cited_paper":"/paper/2505.16113","citing_paper":"/paper/2606.18467"},"observation_digest":"sha256:773e923451d46bff0a469f5b2ead60492d02b145366a2eb495e17eec43032e5a","observation_id":"18f9caa4-06c0-4e9a-8547-b00f52a419ed","resolution":{"observed_at":"2026-07-03T23:29:02.505043Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16113","snapshot_observed_at":"2026-07-12T06:24:52.086585Z","title":"Tools in the loop: Quantifying uncertainty of llm question answering systems that use tools,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02882","last_updated":"2026-07-03T02:28:32Z","snapshot_observed_at":"2026-08-01T20:07:04.724562Z","submitted_at":"2026-07-03T02:28:32Z","title":"Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-12T06:24:52.086585Z"},"links":{"cited_paper":"/paper/2505.16113","citing_paper":"/paper/2607.02882"},"observation_digest":"sha256:5b5ffe3198a025fc3b12106e88bfa0348f09b5ad46c82413ed175b01a8b53ab9","observation_id":"a9b73959-b485-4a26-b3ed-ff9adbc85a61","resolution":{"observed_at":"2026-07-12T06:24:52.086585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16113","snapshot_observed_at":"2026-08-01T08:46:14.378320Z","title":"Tools in the loop: Quantifying uncertainty of llm question answering systems that use tools","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21010","last_updated":"2026-07-23T07:48:20Z","snapshot_observed_at":"2026-08-03T06:18:51.609997Z","submitted_at":"2026-07-23T07:48:20Z","title":"Reexamining zero-shot summarization: Empirical investigation of trustworthiness of LLM-summarizers","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T08:46:14.378320Z"},"links":{"cited_paper":"/paper/2505.16113","citing_paper":"/paper/2607.21010"},"observation_digest":"sha256:adc506237b9462a18cfbbb4543e8bbdab7eea56ad97e337c954e331e337ed210","observation_id":"10f62dba-8f8e-4273-a3e8-f9a326ee90f8","resolution":{"observed_at":"2026-08-01T08:46:14.378320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.16113/citation-record","integrity":"/paper/2505.16113/integrity","json":"/paper/2505.16113/citation-record.json","paper":"/paper/2505.16113"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2211.12717","last_updated":"2022-11-23T05:44:42Z","snapshot_observed_at":"2026-08-03T17:29:08.095061Z","submitted_at":"2022-11-23T05:44:42Z","title":"Benchmarking Bayesian Deep Learning on Diabetic Retinopathy Detection Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.12717","snapshot_observed_at":"2026-08-07T15:11:45.024634Z","title":"Benchmarking bayesian deep learning on diabetic retinopathy detection tasks","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.024634Z"},"links":{"cited_paper":"/paper/2211.12717","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:9500c53337447aec266f1396a1fe29ff68b49cb9a41345ba3b8471063c8084fc","observation_id":"025f2365-18a6-4afb-a3c6-37b30ae1d0e9","resolution":{"observed_at":"2026-08-07T15:11:45.024634Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:48.449007Z","title":"Boolq: Exploring the surprising difficulty of natural yes/no ques- tions","venue":null,"work_id":"91184ac6-0832-4105-89e4-ae4f965ed7ea","year":2019},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.065298Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:094bf2865ceb9d26372a004d0b5a46d81ab522f64bc144ec5d91d56b39f21603","observation_id":"3e90927a-26d4-4e50-88ac-cace3e22a086","resolution":{"observed_at":"2026-08-07T15:11:48.530797Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:48.157099Z","title":"Detecting hallucina- tions in large language models using semantic entropy","venue":null,"work_id":"95704b42-5537-49d4-951e-eba3edab7342","year":2024},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.141259Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:52eaaba5d3b3d65d14d761fea82da797e6860d3e7913db8089890754fe02a74a","observation_id":"b5589958-8520-4e0c-87cb-fe19a46bdc04","resolution":{"observed_at":"2026-08-07T15:11:48.349186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:45.244890Z","title":null,"venue":null,"work_id":null,"year":1936},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.244890Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:7b539e8acc06e46041d26b2e147dea79db203a032943da73383a98ecfb82fe1c","observation_id":"4ce3ef00-44c0-4115-8dc4-6964fd2cbedf","resolution":{"observed_at":"2026-08-07T15:11:45.244890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:48.002704Z","title":"Dropout as a bayesian approximation: Representing model uncertainty in deep learning","venue":null,"work_id":"96db778b-bd69-4b4f-ba10-1856a79c7ef5","year":2016},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.455818Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:4fef6067b7f32112e90a6ac69d0630a0adb9cfe5ff15db1f7fff1802ec73a6b8","observation_id":"9b0a6620-f0d5-4f18-abaa-cbe46a657613","resolution":{"observed_at":"2026-08-07T15:11:48.038983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10997","last_updated":"2024-03-27T09:16:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-18T07:47:33Z","title":"Retrieval-Augmented Generation for Large Language Models: A Survey","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.10997","snapshot_observed_at":"2026-08-07T15:11:45.611899Z","title":"Retrieval-augmented 11 generation for large language models: A sur- vey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.611899Z"},"links":{"cited_paper":"/paper/2312.10997","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:d7e4f8e3ffcd558aad88a00f840405711e2db388918bb407a03b5ab4b7b33415","observation_id":"ca60c6f2-bc36-483a-a178-859e90297ca2","resolution":{"observed_at":"2026-08-07T15:11:45.611899Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2109.13916","last_updated":"2022-06-16T21:12:42Z","snapshot_observed_at":"2026-07-06T11:52:19.105210Z","submitted_at":"2021-09-28T17:59:36Z","title":"Unsolved Problems in ML Safety","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2109.13916","snapshot_observed_at":"2026-08-07T15:11:45.769975Z","title":"Unsolved problems in ml safety","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.769975Z"},"links":{"cited_paper":"/paper/2109.13916","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:e1b20ebd88a7e39811a9f4a7965930688b766412b7e5f4c4146b10d44daa29a9","observation_id":"f774ab00-ed1c-437b-a2db-8fe632399df9","resolution":{"observed_at":"2026-08-07T15:11:45.769975Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.09664","last_updated":"2023-04-15T12:55:45Z","snapshot_observed_at":"2026-07-06T14:53:27.667483Z","submitted_at":"2023-02-19T20:10:07Z","title":"Semantic Uncertainty: Linguistic Invariances for Uncertainty Estimation in Natural Language Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.09664","snapshot_observed_at":"2026-08-07T15:11:45.904860Z","title":"Semantic uncertainty: Linguistic in- variances for uncertainty estimation in nat- ural language generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:45.904860Z"},"links":{"cited_paper":"/paper/2302.09664","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:9cc743b536fa4e0f5f2da118b8b2f4f0fafb7178b316d1739456503517e7f99e","observation_id":"c975e134-c24e-47c9-b900-1294177dc146","resolution":{"observed_at":"2026-08-07T15:11:45.904860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03718","last_updated":"2023-11-26T00:48:12Z","snapshot_observed_at":"2026-08-06T21:35:45.212216Z","submitted_at":"2023-11-26T00:48:12Z","title":"Large Language Models in Law: A Survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03718","snapshot_observed_at":"2026-08-07T15:11:46.036999Z","title":"Large language models in law: A survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.036999Z"},"links":{"cited_paper":"/paper/2312.03718","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:0f7373e06571aec3687b48a4c7161de5538544b0de9da4cff94cfa017d05edc6","observation_id":"5828cb78-9767-4de0-8c0d-3e3500a323a1","resolution":{"observed_at":"2026-08-07T15:11:46.036999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.852172Z","title":"Simple and scalable pre- dictive uncertainty estimation using deep en- sembles","venue":null,"work_id":"b92fed0e-fbc4-43bc-b79b-ec8bd26a7262","year":2017},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.093367Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:b9751f4cc315d95efeef288a04703274d3ccbef22906d9ae70f82cac4379b196","observation_id":"7b62fc94-35ec-4a67-a773-6b93691d7c96","resolution":{"observed_at":"2026-08-07T15:11:47.965482Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.722796Z","title":"Retrieval-augmented generation for knowledge-intensive nlp tasks","venue":null,"work_id":"c8820b6d-9900-4b96-88d6-486ada3f7df1","year":2020},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.141626Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:10813ecfe92fa341312503a8126bd84df41773407475a079ec2b759e6e9191ca","observation_id":"b76735cd-c92d-4c4a-acb3-aeb5b18021c5","resolution":{"observed_at":"2026-08-07T15:11:47.764858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.15993","last_updated":"2024-10-23T08:33:54Z","snapshot_observed_at":"2026-08-08T05:41:54.254097Z","submitted_at":"2024-04-24T17:10:35Z","title":"Uncertainty Estimation and Quantification for LLMs: A Simple Supervised Approach","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.15993","snapshot_observed_at":"2026-08-07T15:11:46.190902Z","title":"Uncertainty estimation and quantifi- cation for llms: A simple supervised approach","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.190902Z"},"links":{"cited_paper":"/paper/2404.15993","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:b5011707244d31a4aaccedf509dcfd04f907033ce3632df5db62c95998b8544d","observation_id":"ce0d25f6-6fa4-404f-ba4b-9f994604945a","resolution":{"observed_at":"2026-08-07T15:11:46.190902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00920","last_updated":"2025-07-25T08:26:54Z","snapshot_observed_at":"2026-08-07T03:58:38.931628Z","submitted_at":"2024-09-02T03:19:56Z","title":"ToolACE: Winning the Points of LLM Function Calling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.00920","snapshot_observed_at":"2026-08-07T15:11:46.258369Z","title":"Toolace: Winning the points of llm func- tion calling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.258369Z"},"links":{"cited_paper":"/paper/2409.00920","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:7839b48183edf2b3190a986fdc3673356cd853cf46ed3013146aca5f59f29ae2","observation_id":"924ee281-803d-45e2-a9b0-863ded1c79fa","resolution":{"observed_at":"2026-08-07T15:11:46.258369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.07650","last_updated":"2021-02-11T09:42:35Z","snapshot_observed_at":"2026-08-03T18:01:42.203617Z","submitted_at":"2020-02-18T15:40:13Z","title":"Uncertainty Estimation in Autoregressive Structured Prediction","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.07650","snapshot_observed_at":"2026-08-07T15:11:46.329808Z","title":"Uncertainty estimation in autoregressive structured predic- tion","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.329808Z"},"links":{"cited_paper":"/paper/2002.07650","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:cee0532886333c629de20d208385b812ff9a8dc2fad2158619923e1c9af4f173","observation_id":"a481744f-c47d-4a40-8368-4cac467687bd","resolution":{"observed_at":"2026-08-07T15:11:46.329808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.601308Z","title":"Revisiting, benchmarking and exploring api recommendation: How far are we?, 2021","venue":null,"work_id":"f425789f-506d-4264-a674-bdda3d12a037","year":2021},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.380342Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:3f11b564ff04c9d6dbc92dfaa4a8b88a712544c2862b55f9a3efff4bea83e5ee","observation_id":"82700843-72b3-487d-b72a-572eccacb412","resolution":{"observed_at":"2026-08-07T15:11:47.667536Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08354","last_updated":"2024-08-06T15:14:36Z","snapshot_observed_at":"2026-08-08T23:22:13.517141Z","submitted_at":"2023-04-17T15:16:10Z","title":"Tool Learning with Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08354","snapshot_observed_at":"2026-08-07T15:11:46.423549Z","title":"Tool learning with foundation models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.423549Z"},"links":{"cited_paper":"/paper/2304.08354","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:4b5495921a8ea847e27f5347aaedab2d383bf86a2ac9e7795d80e30f1d52a47b","observation_id":"77667ac6-9d8f-4b70-8998-c3723011a061","resolution":{"observed_at":"2026-08-07T15:11:46.423549Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17935","last_updated":"2024-11-04T15:07:18Z","snapshot_observed_at":"2026-08-09T04:01:18.209494Z","submitted_at":"2024-05-28T08:01:26Z","title":"Tool Learning with Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17935","snapshot_observed_at":"2026-08-07T15:11:46.515086Z","title":"Tool learning with large language models: A survey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.515086Z"},"links":{"cited_paper":"/paper/2405.17935","citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:55d8e4101d5b5941c0fcbe3dd47d862485df576bfe214db5eff70d226b1e695f","observation_id":"0b188d76-8f2e-4a63-b6bf-411e9f8cd342","resolution":{"observed_at":"2026-08-07T15:11:46.515086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.464064Z","title":"Toolformer: Language models can teach themselves to use tools","venue":null,"work_id":"a6f78e0c-83d2-4971-9dc2-c311bfc11188","year":2023},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.583746Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:15ed62bf0f2eec1c01aa8f266a9add55705fc68bba2d7bcf8202d9c39309b6f9","observation_id":"b61fa36e-31df-4c1e-921f-ee98d4dfafdc","resolution":{"observed_at":"2026-08-07T15:11:47.495778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.305716Z","title":"Using the adap learning algorithm to forecast the onset of diabetes mellitus","venue":null,"work_id":"40493468-22e8-4773-b5d7-a85c179c11e5","year":1988},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.615512Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:3753ee46db2fc9ee86d58504f10f1487e1d53f33e409d25be18ed18052989469","observation_id":"9e6f5e98-a010-40b5-9936-c4bcbb102d38","resolution":{"observed_at":"2026-08-07T15:11:47.379534Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.197483Z","title":"Question generation as a competitive undergraduate course project","venue":null,"work_id":"c961bd7e-d3e3-44f4-bf56-51bd845435aa","year":2008},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.677222Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:5d9b54855f8803072213482753dcac6a42c769e6a534383a967af999ee628865","observation_id":"dfe20a7a-299d-456e-abd6-4c2c6bfd6d07","resolution":{"observed_at":"2026-08-07T15:11:47.248402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:46.763294Z","title":"Large language models in medicine","venue":null,"work_id":null,"year":1930},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.763294Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:dc5ff65824caca1600f99fb2cf92a08fbdde593b4f0fe84610fac1e8bbd7a206","observation_id":"0a38c0fe-5290-47c4-9a3c-13e795e38be4","resolution":{"observed_at":"2026-08-07T15:11:46.763294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T15:11:47.052249Z","title":"Toolqa: A dataset for 12 llm question answering with external tools","venue":null,"work_id":"5acef799-06a9-4c7b-be4a-329a4a21fc6a","year":2023},"citing_paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T15:11:46.806710Z"},"links":{"citing_paper":"/paper/2505.16113"},"observation_digest":"sha256:faf3167ce6faae01619fcad81691d6f311eb5348e58bfd9f69702f79e2bc30a0","observation_id":"a17d443b-48c9-454c-a8a3-2e5785c0ee5c","resolution":{"observed_at":"2026-08-07T15:11:47.080564Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.16113","last_updated":"2025-05-22T01:34:23Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-08T12:58:05.133406Z","submitted_at":"2025-05-22T01:34:23Z","title":"Tools in the Loop: Quantifying Uncertainty of LLM Question Answering Systems That Use Tools"},"reference_resolution":{"displayed":22,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":0,"verified_fuzzy":10},"total_outbound_references":22},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 22 of 22 outbound references and 4 inbound Pith citation observations for arXiv:2505.16113."}