{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HOELECPDLT74JY3YSBKYS23KZN","short_pith_number":"pith:HOELECPD","schema_version":"1.0","canonical_sha256":"3b88b209e35cffc4e3789055896b6acb482581d7ebec57cadb71d65b4b3500fb","source":{"kind":"arxiv","id":"2507.01489","version":1},"attestation_state":"computed","paper":{"title":"Agent-as-Tool: A Study on the Hierarchical Decision Making with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.AI","authors_text":"Yanfei Zhang","submitted_at":"2025-07-02T08:49:43Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as one of the most significant technological advancements in artificial intelligence in recent years. Their ability to understand, generate, and reason with natural language has transformed how we interact with AI systems. With the development of LLM-based agents and reinforcement-learning-based reasoning models, the study of applying reinforcement learning in agent frameworks has become a new research focus. However, all previous studies face the challenge of deciding the tool calling process and the reasoning process simultaneously, and the chain of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.01489","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-02T08:49:43Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"9c013976575c81a2fa74fbfc04f203363f7bd4b75a65a5903895fee130993ddb","abstract_canon_sha256":"5ec7cd1895c71276593f0fadc82b9d5e36fd5a85a96da7afeb589d31b8c7b096"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:49.808696Z","signature_b64":"YudJ1S+kLzzUngB0Cg9ct9fU31K5YrtBGQz2mpfP2F6HsS+Ak5GF2CIX2Z1pgpNJFNpqa7WEokCLVN+UfNHWAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b88b209e35cffc4e3789055896b6acb482581d7ebec57cadb71d65b4b3500fb","last_reissued_at":"2026-07-05T11:30:49.808217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:49.808217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Agent-as-Tool: A Study on the Hierarchical Decision Making with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.AI","authors_text":"Yanfei Zhang","submitted_at":"2025-07-02T08:49:43Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as one of the most significant technological advancements in artificial intelligence in recent years. Their ability to understand, generate, and reason with natural language has transformed how we interact with AI systems. With the development of LLM-based agents and reinforcement-learning-based reasoning models, the study of applying reinforcement learning in agent frameworks has become a new research focus. However, all previous studies face the challenge of deciding the tool calling process and the reasoning process simultaneously, and the chain of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.01489","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.01489/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.01489","created_at":"2026-07-05T11:30:49.808276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.01489v1","created_at":"2026-07-05T11:30:49.808276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.01489","created_at":"2026-07-05T11:30:49.808276+00:00"},{"alias_kind":"pith_short_12","alias_value":"HOELECPDLT74","created_at":"2026-07-05T11:30:49.808276+00:00"},{"alias_kind":"pith_short_16","alias_value":"HOELECPDLT74JY3Y","created_at":"2026-07-05T11:30:49.808276+00:00"},{"alias_kind":"pith_short_8","alias_value":"HOELECPD","created_at":"2026-07-05T11:30:49.808276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27788","citing_title":"Knowing When to Ask: Segment-Level Credit Assignment for LLM Tool Use","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22154","citing_title":"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11853","citing_title":"GEAR: Granularity-Adaptive Advantage Reweighting for LLM Agents via Self-Distillation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20857","citing_title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11853","citing_title":"GEAR: Granularity-Adaptive Advantage Reweighting for LLM Agents via Self-Distillation","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN","json":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN.json","graph_json":"https://pith.science/api/pith-number/HOELECPDLT74JY3YSBKYS23KZN/graph.json","events_json":"https://pith.science/api/pith-number/HOELECPDLT74JY3YSBKYS23KZN/events.json","paper":"https://pith.science/paper/HOELECPD"},"agent_actions":{"view_html":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN","download_json":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN.json","view_paper":"https://pith.science/paper/HOELECPD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.01489&json=true","fetch_graph":"https://pith.science/api/pith-number/HOELECPDLT74JY3YSBKYS23KZN/graph.json","fetch_events":"https://pith.science/api/pith-number/HOELECPDLT74JY3YSBKYS23KZN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN/action/storage_attestation","attest_author":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN/action/author_attestation","sign_citation":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN/action/citation_signature","submit_replication":"https://pith.science/pith/HOELECPDLT74JY3YSBKYS23KZN/action/replication_record"}},"created_at":"2026-07-05T11:30:49.808276+00:00","updated_at":"2026-07-05T11:30:49.808276+00:00"}