{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:53B2XUSXTAI6VENF5SH432EBEO","short_pith_number":"pith:53B2XUSX","canonical_record":{"source":{"id":"2605.00737","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-01T15:38:13Z","cross_cats_sorted":[],"title_canon_sha256":"0f33d388722aee89912de8e63e90f6ad1289ecdda749845883f04934c5b2cd16","abstract_canon_sha256":"c668fcd8094d98646cbbf9665fb0b6d8071744882c162ae7d46858cd577a92c4"},"schema_version":"1.0"},"canonical_sha256":"eec3abd2579811ea91a5ec8fcde881239ff18dd71ed97c77ab4215f0af386388","source":{"kind":"arxiv","id":"2605.00737","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.00737","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"arxiv_version","alias_value":"2605.00737v2","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.00737","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"pith_short_12","alias_value":"53B2XUSXTAI6","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"pith_short_16","alias_value":"53B2XUSXTAI6VENF","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"pith_short_8","alias_value":"53B2XUSX","created_at":"2026-06-08T01:04:06Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:53B2XUSXTAI6VENF5SH432EBEO","target":"record","payload":{"canonical_record":{"source":{"id":"2605.00737","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-01T15:38:13Z","cross_cats_sorted":[],"title_canon_sha256":"0f33d388722aee89912de8e63e90f6ad1289ecdda749845883f04934c5b2cd16","abstract_canon_sha256":"c668fcd8094d98646cbbf9665fb0b6d8071744882c162ae7d46858cd577a92c4"},"schema_version":"1.0"},"canonical_sha256":"eec3abd2579811ea91a5ec8fcde881239ff18dd71ed97c77ab4215f0af386388","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-08T01:04:06.264972Z","signature_b64":"0uCvEjkNShQq/94oOOrnqvV5oRPrAnE/HHmsGTONNTqo9UMw04MD7ihdtgYeeqzmiHjUdb/A2IbI0M2pKsXPBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eec3abd2579811ea91a5ec8fcde881239ff18dd71ed97c77ab4215f0af386388","last_reissued_at":"2026-06-08T01:04:06.264125Z","signature_status":"signed_v1","first_computed_at":"2026-06-08T01:04:06.264125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.00737","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-08T01:04:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zAe1HGW7G3xI37GAvb1Mo/PUipb7SM5bsqysQquYJ8BuqBAUleXfCjDKdGzeOuw3JQiCg9Pe2C8lp9whuCabCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T22:05:51.065267Z"},"content_sha256":"14e6df034cde74c2163bfab88d89d91070eedcebefaee6c77b772d33a9570c52","schema_version":"1.0","event_id":"sha256:14e6df034cde74c2163bfab88d89d91070eedcebefaee6c77b772d33a9570c52"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:53B2XUSXTAI6VENF5SH432EBEO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"To Call or Not to Call: A Framework to Assess and Optimize LLM Tool Calling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"LLMs often misjudge when calling tools like web search is truly necessary or useful, but estimators built from their internal hidden states can make better calls and raise task performance.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Abhilasha Ravichander, Arijit Nag, Krishna P. Gummadi, Mahsa Amani, Muhammad Bilal Zafar, Qinyuan Wu, Seungeon Lee, Soumi Das","submitted_at":"2026-05-01T15:38:13Z","abstract_excerpt":"Agentic AI architectures augment LLMs with external tools, unlocking strong capabilities. However, tool use is not always beneficial; some calls may be redundant or even harmful. Effective tool use, therefore, hinges on a core LLM decision: whether to call or not call a tool when performing a task. This decision is particularly challenging for web search tools, where the benefits of external information depend on the model's internal knowledge and its ability to integrate potentially noisy tool responses. We introduce a principled framework inspired by decision-making theory to evaluate web se"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Models' perceived need and utility of tool calls are often misaligned with their true need and utility. Building on this framework, we train lightweight estimators of need and utility based on models' hidden states. Our estimators enable simple controllers that can improve decision quality and lead to stronger task performance than the self-perceived set up across three tasks and six models.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That true need and utility can be reliably inferred from an optimal allocation of tool calls to serve as ground truth for training the estimators, and that this inference generalizes across tasks without introducing selection bias.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"LLMs often misalign their self-perceived need for tools with true need and utility, but lightweight estimators trained on hidden states can improve tool-calling decisions and task performance across multiple models and tasks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"LLMs often misjudge when calling tools like web search is truly necessary or useful, but estimators built from their internal hidden states can make better calls and raise task performance.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"64a0dc0f297b47b6427a5ab0a82d51a98c25721e9852ee91ec4352f814a62058"},"source":{"id":"2605.00737","kind":"arxiv","version":2},"verdict":{"id":"3deb2aa3-8859-44fe-a381-c4083c7e8a42","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-09T19:29:37.035402Z","strongest_claim":"Models' perceived need and utility of tool calls are often misaligned with their true need and utility. Building on this framework, we train lightweight estimators of need and utility based on models' hidden states. Our estimators enable simple controllers that can improve decision quality and lead to stronger task performance than the self-perceived set up across three tasks and six models.","one_line_summary":"LLMs often misalign their self-perceived need for tools with true need and utility, but lightweight estimators trained on hidden states can improve tool-calling decisions and task performance across multiple models and tasks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That true need and utility can be reliably inferred from an optimal allocation of tool calls to serve as ground truth for training the estimators, and that this inference generalizes across tasks without introducing selection bias.","pith_extraction_headline":"LLMs often misjudge when calling tools like web search is truly necessary or useful, but estimators built from their internal hidden states can make better calls and raise task performance."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.00737/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-20T19:35:37.927837Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T17:53:16.363230Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"352eb1b0ce49c8841fb7edd33d3e1c116dbd99a4a845dd82f500ec763fe7d847"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"3deb2aa3-8859-44fe-a381-c4083c7e8a42"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-06-08T01:04:06Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rEx9W9ODiRqx7cT144kSezUtxwXZvrY7IMbEgzPhbw5QsP9cj8t/xSlsorzvMMS/a8qYMHaXukTRlgoNVedkCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-22T22:05:51.065743Z"},"content_sha256":"dd292698fff86fecdfdef7e3acb7a2c23d643dc30ae4776c26fe145b2e3015ac","schema_version":"1.0","event_id":"sha256:dd292698fff86fecdfdef7e3acb7a2c23d643dc30ae4776c26fe145b2e3015ac"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/53B2XUSXTAI6VENF5SH432EBEO/bundle.json","state_url":"https://pith.science/pith/53B2XUSXTAI6VENF5SH432EBEO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/53B2XUSXTAI6VENF5SH432EBEO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-22T22:05:51Z","links":{"resolver":"https://pith.science/pith/53B2XUSXTAI6VENF5SH432EBEO","bundle":"https://pith.science/pith/53B2XUSXTAI6VENF5SH432EBEO/bundle.json","state":"https://pith.science/pith/53B2XUSXTAI6VENF5SH432EBEO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/53B2XUSXTAI6VENF5SH432EBEO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:53B2XUSXTAI6VENF5SH432EBEO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c668fcd8094d98646cbbf9665fb0b6d8071744882c162ae7d46858cd577a92c4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-01T15:38:13Z","title_canon_sha256":"0f33d388722aee89912de8e63e90f6ad1289ecdda749845883f04934c5b2cd16"},"schema_version":"1.0","source":{"id":"2605.00737","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.00737","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"arxiv_version","alias_value":"2605.00737v2","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.00737","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"pith_short_12","alias_value":"53B2XUSXTAI6","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"pith_short_16","alias_value":"53B2XUSXTAI6VENF","created_at":"2026-06-08T01:04:06Z"},{"alias_kind":"pith_short_8","alias_value":"53B2XUSX","created_at":"2026-06-08T01:04:06Z"}],"graph_snapshots":[{"event_id":"sha256:dd292698fff86fecdfdef7e3acb7a2c23d643dc30ae4776c26fe145b2e3015ac","target":"graph","created_at":"2026-06-08T01:04:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Models' perceived need and utility of tool calls are often misaligned with their true need and utility. Building on this framework, we train lightweight estimators of need and utility based on models' hidden states. Our estimators enable simple controllers that can improve decision quality and lead to stronger task performance than the self-perceived set up across three tasks and six models."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That true need and utility can be reliably inferred from an optimal allocation of tool calls to serve as ground truth for training the estimators, and that this inference generalizes across tasks without introducing selection bias."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"LLMs often misalign their self-perceived need for tools with true need and utility, but lightweight estimators trained on hidden states can improve tool-calling decisions and task performance across multiple models and tasks."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"LLMs often misjudge when calling tools like web search is truly necessary or useful, but estimators built from their internal hidden states can make better calls and raise task performance."}],"snapshot_sha256":"64a0dc0f297b47b6427a5ab0a82d51a98c25721e9852ee91ec4352f814a62058"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T19:35:37.927837Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T17:53:16.363230Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.00737/integrity.json","findings":[],"snapshot_sha256":"352eb1b0ce49c8841fb7edd33d3e1c116dbd99a4a845dd82f500ec763fe7d847","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Agentic AI architectures augment LLMs with external tools, unlocking strong capabilities. However, tool use is not always beneficial; some calls may be redundant or even harmful. Effective tool use, therefore, hinges on a core LLM decision: whether to call or not call a tool when performing a task. This decision is particularly challenging for web search tools, where the benefits of external information depend on the model's internal knowledge and its ability to integrate potentially noisy tool responses. We introduce a principled framework inspired by decision-making theory to evaluate web se","authors_text":"Abhilasha Ravichander, Arijit Nag, Krishna P. Gummadi, Mahsa Amani, Muhammad Bilal Zafar, Qinyuan Wu, Seungeon Lee, Soumi Das","cross_cats":[],"headline":"LLMs often misjudge when calling tools like web search is truly necessary or useful, but estimators built from their internal hidden states can make better calls and raise task performance.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-01T15:38:13Z","title":"To Call or Not to Call: A Framework to Assess and Optimize LLM Tool Calling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.00737","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-09T19:29:37.035402Z","id":"3deb2aa3-8859-44fe-a381-c4083c7e8a42","model_set":{"reader":"grok-4.3"},"one_line_summary":"LLMs often misalign their self-perceived need for tools with true need and utility, but lightweight estimators trained on hidden states can improve tool-calling decisions and task performance across multiple models and tasks.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"LLMs often misjudge when calling tools like web search is truly necessary or useful, but estimators built from their internal hidden states can make better calls and raise task performance.","strongest_claim":"Models' perceived need and utility of tool calls are often misaligned with their true need and utility. Building on this framework, we train lightweight estimators of need and utility based on models' hidden states. Our estimators enable simple controllers that can improve decision quality and lead to stronger task performance than the self-perceived set up across three tasks and six models.","weakest_assumption":"That true need and utility can be reliably inferred from an optimal allocation of tool calls to serve as ground truth for training the estimators, and that this inference generalizes across tasks without introducing selection bias."}},"verdict_id":"3deb2aa3-8859-44fe-a381-c4083c7e8a42"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:14e6df034cde74c2163bfab88d89d91070eedcebefaee6c77b772d33a9570c52","target":"record","created_at":"2026-06-08T01:04:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c668fcd8094d98646cbbf9665fb0b6d8071744882c162ae7d46858cd577a92c4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-05-01T15:38:13Z","title_canon_sha256":"0f33d388722aee89912de8e63e90f6ad1289ecdda749845883f04934c5b2cd16"},"schema_version":"1.0","source":{"id":"2605.00737","kind":"arxiv","version":2}},"canonical_sha256":"eec3abd2579811ea91a5ec8fcde881239ff18dd71ed97c77ab4215f0af386388","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"eec3abd2579811ea91a5ec8fcde881239ff18dd71ed97c77ab4215f0af386388","first_computed_at":"2026-06-08T01:04:06.264125Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-08T01:04:06.264125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"0uCvEjkNShQq/94oOOrnqvV5oRPrAnE/HHmsGTONNTqo9UMw04MD7ihdtgYeeqzmiHjUdb/A2IbI0M2pKsXPBw==","signature_status":"signed_v1","signed_at":"2026-06-08T01:04:06.264972Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.00737","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:14e6df034cde74c2163bfab88d89d91070eedcebefaee6c77b772d33a9570c52","sha256:dd292698fff86fecdfdef7e3acb7a2c23d643dc30ae4776c26fe145b2e3015ac"],"state_sha256":"061e6c2f165f76e6f794096eb7867679cc3972abeee1c8badfe0df04cca14012"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Ge7pvL63UjlME4ebysdrVX7eu2QDp8TWRnyTe2djqT0I7DGhmF8z1wiZi62DYdJFDKv/N2/Jok+IzNj9+zVnDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-22T22:05:51.068425Z","bundle_sha256":"0d5cfc9c3292b72d11f68fa05cd09d04f79546ca26b3d580e458ab213ba205e9"}}