{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:UA5SYVJDAWARPFT75HIRHY5VPC","short_pith_number":"pith:UA5SYVJD","schema_version":"1.0","canonical_sha256":"a03b2c5523058117967fe9d113e3b578b3b78d604cbc74cb9b1b00ee4f19873c","source":{"kind":"arxiv","id":"2608.01321","version":1},"attestation_state":"computed","paper":{"title":"BiCAA: Bidirectional Credit Assignment for Search-Augmented Agent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bin Xu, Conghui Zhu, Hailong Cao, Yibin Huang","submitted_at":"2026-08-02T15:41:57Z","abstract_excerpt":"Multi-step search is a fundamental capability for search agents, enabling them to iteratively acquire, refine, and integrate external evidence for complex reasoning QA. However, vanilla GRPO allocates rewards exclusively based on the model's final outputs, yielding outcome-only supervision with no supervisory signals for intermediate reasoning steps. Such sparse supervision easily causes training instability and redundant search behaviors on multi-step search tasks. To mitigate this limitation, we adopt process reward to deliver stepwise supervision signals. For this process reward, we propose"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.01321","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-02T15:41:57Z","cross_cats_sorted":[],"title_canon_sha256":"3a299da183a16a56d21f104ca1c1f048f87272d15c32d7478cd6b9ebf9448ef1","abstract_canon_sha256":"6dc3779aa3ad2db87b51928b0b862a0463490cbf253dde86d2b6b95ee73899ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T02:04:01.671346Z","signature_b64":"80rl8O8eKYRYFwUvlgwxVpovLbfaeevDDVXb8FUPqtM7f0yvaMD1wCPF5/LWonVKYpa4gm/HS9vLZ10MF3saBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a03b2c5523058117967fe9d113e3b578b3b78d604cbc74cb9b1b00ee4f19873c","last_reissued_at":"2026-08-04T02:04:01.665110Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T02:04:01.665110Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BiCAA: Bidirectional Credit Assignment for Search-Augmented Agent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bin Xu, Conghui Zhu, Hailong Cao, Yibin Huang","submitted_at":"2026-08-02T15:41:57Z","abstract_excerpt":"Multi-step search is a fundamental capability for search agents, enabling them to iteratively acquire, refine, and integrate external evidence for complex reasoning QA. However, vanilla GRPO allocates rewards exclusively based on the model's final outputs, yielding outcome-only supervision with no supervisory signals for intermediate reasoning steps. Such sparse supervision easily causes training instability and redundant search behaviors on multi-step search tasks. To mitigate this limitation, we adopt process reward to deliver stepwise supervision signals. For this process reward, we propose"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.01321","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.01321/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.01321","created_at":"2026-08-04T02:04:01.669472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.01321v1","created_at":"2026-08-04T02:04:01.669472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.01321","created_at":"2026-08-04T02:04:01.669472+00:00"},{"alias_kind":"pith_short_12","alias_value":"UA5SYVJDAWAR","created_at":"2026-08-04T02:04:01.669472+00:00"},{"alias_kind":"pith_short_16","alias_value":"UA5SYVJDAWARPFT7","created_at":"2026-08-04T02:04:01.669472+00:00"},{"alias_kind":"pith_short_8","alias_value":"UA5SYVJD","created_at":"2026-08-04T02:04:01.669472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC","json":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC.json","graph_json":"https://pith.science/api/pith-number/UA5SYVJDAWARPFT75HIRHY5VPC/graph.json","events_json":"https://pith.science/api/pith-number/UA5SYVJDAWARPFT75HIRHY5VPC/events.json","paper":"https://pith.science/paper/UA5SYVJD"},"agent_actions":{"view_html":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC","download_json":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC.json","view_paper":"https://pith.science/paper/UA5SYVJD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.01321&json=true","fetch_graph":"https://pith.science/api/pith-number/UA5SYVJDAWARPFT75HIRHY5VPC/graph.json","fetch_events":"https://pith.science/api/pith-number/UA5SYVJDAWARPFT75HIRHY5VPC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC/action/storage_attestation","attest_author":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC/action/author_attestation","sign_citation":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC/action/citation_signature","submit_replication":"https://pith.science/pith/UA5SYVJDAWARPFT75HIRHY5VPC/action/replication_record"}},"created_at":"2026-08-04T02:04:01.669472+00:00","updated_at":"2026-08-04T02:04:01.669472+00:00"}