{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FADONVWV2AVLZAY5J3BKJBARBQ","short_pith_number":"pith:FADONVWV","schema_version":"1.0","canonical_sha256":"2806e6d6d5d02abc831d4ec2a484110c066601c9b61c28eff0e431bbf8127f90","source":{"kind":"arxiv","id":"2510.19771","version":4},"attestation_state":"computed","paper":{"title":"Beyond Reactivity: Measuring Proactive Problem Solving in LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ash Lewis, Dheeraj Rajagopal, Dhruv Atreja, George Hurn-Maloney, Gil Pasternak, Julia White, Matthew Thomas","submitted_at":"2025-10-22T17:00:45Z","abstract_excerpt":"LLM-based agents are increasingly moving towards proactivity: rather than awaiting instruction, they exercise agency to anticipate user needs and solve them autonomously. However, evaluating proactivity is challenging; current benchmarks are constrained to localized context, limiting their ability to test reasoning across sources and longer time horizons. To address this gap, we present PROBE (Proactive Resolution Of BottlEnecks). PROBE decomposes proactivity as a pipeline of three core capabilities: (1) searching for unspecified issues, (2) identifying specific bottlenecks, and (3) executing "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.19771","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-10-22T17:00:45Z","cross_cats_sorted":[],"title_canon_sha256":"b62215d0640efc2c6666ce9b84695cf81cb8ae5f33c2d0b47d726f1ad0577962","abstract_canon_sha256":"f008c4a5e6d70f5fd758bcc91b278fdca047cf47e829d8919546cd263bc98de5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-08T01:19:07.853471Z","signature_b64":"jO6RUZ1skJpHi6JAxz1b9qKvoa1tTL5ikDnIr2p6nDbYms5DUQrlTmOAzL3DGsi+DsoskrjVcx8sAkeCkhRIDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2806e6d6d5d02abc831d4ec2a484110c066601c9b61c28eff0e431bbf8127f90","last_reissued_at":"2026-07-08T01:19:07.852956Z","signature_status":"signed_v1","first_computed_at":"2026-07-08T01:19:07.852956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Reactivity: Measuring Proactive Problem Solving in LLM Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ash Lewis, Dheeraj Rajagopal, Dhruv Atreja, George Hurn-Maloney, Gil Pasternak, Julia White, Matthew Thomas","submitted_at":"2025-10-22T17:00:45Z","abstract_excerpt":"LLM-based agents are increasingly moving towards proactivity: rather than awaiting instruction, they exercise agency to anticipate user needs and solve them autonomously. However, evaluating proactivity is challenging; current benchmarks are constrained to localized context, limiting their ability to test reasoning across sources and longer time horizons. To address this gap, we present PROBE (Proactive Resolution Of BottlEnecks). PROBE decomposes proactivity as a pipeline of three core capabilities: (1) searching for unspecified issues, (2) identifying specific bottlenecks, and (3) executing "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.19771","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.19771/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.19771","created_at":"2026-07-08T01:19:07.853013+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.19771v4","created_at":"2026-07-08T01:19:07.853013+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.19771","created_at":"2026-07-08T01:19:07.853013+00:00"},{"alias_kind":"pith_short_12","alias_value":"FADONVWV2AVL","created_at":"2026-07-08T01:19:07.853013+00:00"},{"alias_kind":"pith_short_16","alias_value":"FADONVWV2AVLZAY5","created_at":"2026-07-08T01:19:07.853013+00:00"},{"alias_kind":"pith_short_8","alias_value":"FADONVWV","created_at":"2026-07-08T01:19:07.853013+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":5,"sample":[{"citing_arxiv_id":"2606.05557","citing_title":"AURA: Intent-Directed Probing for Implicit-Need Surfacing in Situated LLM Agents","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04743","citing_title":"TIDE: Proactive Multi-Problem Discovery via Template-Guided Iteration","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2605.09228","citing_title":"ProactBench: Beyond What The User Asked For","ref_index":136,"is_internal_anchor":true},{"citing_arxiv_id":"2604.14228","citing_title":"Dive into Claude Code: The Design Space of Today's and Future AI Agent Systems","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2605.06717","citing_title":"Agentic Coding Needs Proactivity, Not Just Autonomy","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ","json":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ.json","graph_json":"https://pith.science/api/pith-number/FADONVWV2AVLZAY5J3BKJBARBQ/graph.json","events_json":"https://pith.science/api/pith-number/FADONVWV2AVLZAY5J3BKJBARBQ/events.json","paper":"https://pith.science/paper/FADONVWV"},"agent_actions":{"view_html":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ","download_json":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ.json","view_paper":"https://pith.science/paper/FADONVWV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.19771&json=true","fetch_graph":"https://pith.science/api/pith-number/FADONVWV2AVLZAY5J3BKJBARBQ/graph.json","fetch_events":"https://pith.science/api/pith-number/FADONVWV2AVLZAY5J3BKJBARBQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ/action/storage_attestation","attest_author":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ/action/author_attestation","sign_citation":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ/action/citation_signature","submit_replication":"https://pith.science/pith/FADONVWV2AVLZAY5J3BKJBARBQ/action/replication_record"}},"created_at":"2026-07-08T01:19:07.853013+00:00","updated_at":"2026-07-08T01:19:07.853013+00:00"}