{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VTTSU3OXZW6Q5VD72HWXJTZUA2","short_pith_number":"pith:VTTSU3OX","schema_version":"1.0","canonical_sha256":"ace72a6dd7cdbd0ed47fd1ed74cf3406b6ab3da96f0c138db744bb6093530d9a","source":{"kind":"arxiv","id":"2309.06553","version":4},"attestation_state":"computed","paper":{"title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alihan H\\\"uy\\\"uk, Hao Sun, Mihaela van der Schaar","submitted_at":"2023-09-13T01:12:52Z","abstract_excerpt":"In this study, we aim to enhance the arithmetic reasoning ability of Large Language Models (LLMs) through zero-shot prompt optimization. We identify a previously overlooked objective of query dependency in such optimization and elucidate two ensuing challenges that impede the successful and economical design of prompt optimization techniques. One primary issue is the absence of an effective method to evaluate prompts during inference when the golden answer is unavailable. Concurrently, learning via interactions with the LLMs to navigate the expansive natural language prompting space proves to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.06553","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-13T01:12:52Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3d53cd5fb52388124f5e44fab647de9dc1fa2b5e93b439ace49974dfb331edb1","abstract_canon_sha256":"6a43712f97a4c09378b7ed93ddb895b83755672a69e02f9bdd117c2d81610e7c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:10.481967Z","signature_b64":"Ro186gjwVrveCJfXSyqR95o/+xSlVUUJlDEiBCf8B3RlYwiHa6RgZIiRrWnSlpLHGx/gDXI2sVcBsOI+YY9yCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ace72a6dd7cdbd0ed47fd1ed74cf3406b6ab3da96f0c138db744bb6093530d9a","last_reissued_at":"2026-07-05T07:53:10.481477Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:10.481477Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Query-Dependent Prompt Evaluation and Optimization with Offline Inverse RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alihan H\\\"uy\\\"uk, Hao Sun, Mihaela van der Schaar","submitted_at":"2023-09-13T01:12:52Z","abstract_excerpt":"In this study, we aim to enhance the arithmetic reasoning ability of Large Language Models (LLMs) through zero-shot prompt optimization. We identify a previously overlooked objective of query dependency in such optimization and elucidate two ensuing challenges that impede the successful and economical design of prompt optimization techniques. One primary issue is the absence of an effective method to evaluate prompts during inference when the golden answer is unavailable. Concurrently, learning via interactions with the LLMs to navigate the expansive natural language prompting space proves to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.06553","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.06553/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.06553","created_at":"2026-07-05T07:53:10.481535+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.06553v4","created_at":"2026-07-05T07:53:10.481535+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.06553","created_at":"2026-07-05T07:53:10.481535+00:00"},{"alias_kind":"pith_short_12","alias_value":"VTTSU3OXZW6Q","created_at":"2026-07-05T07:53:10.481535+00:00"},{"alias_kind":"pith_short_16","alias_value":"VTTSU3OXZW6Q5VD7","created_at":"2026-07-05T07:53:10.481535+00:00"},{"alias_kind":"pith_short_8","alias_value":"VTTSU3OX","created_at":"2026-07-05T07:53:10.481535+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24004","citing_title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24004","citing_title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20727","citing_title":"Supplement Generation Training for Enhancing Agentic Task Performance","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2","json":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2.json","graph_json":"https://pith.science/api/pith-number/VTTSU3OXZW6Q5VD72HWXJTZUA2/graph.json","events_json":"https://pith.science/api/pith-number/VTTSU3OXZW6Q5VD72HWXJTZUA2/events.json","paper":"https://pith.science/paper/VTTSU3OX"},"agent_actions":{"view_html":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2","download_json":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2.json","view_paper":"https://pith.science/paper/VTTSU3OX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.06553&json=true","fetch_graph":"https://pith.science/api/pith-number/VTTSU3OXZW6Q5VD72HWXJTZUA2/graph.json","fetch_events":"https://pith.science/api/pith-number/VTTSU3OXZW6Q5VD72HWXJTZUA2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2/action/storage_attestation","attest_author":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2/action/author_attestation","sign_citation":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2/action/citation_signature","submit_replication":"https://pith.science/pith/VTTSU3OXZW6Q5VD72HWXJTZUA2/action/replication_record"}},"created_at":"2026-07-05T07:53:10.481535+00:00","updated_at":"2026-07-05T07:53:10.481535+00:00"}