{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:QR556MGWIBCLG3HC7PTTDE4AZ2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"512f14c2915de54d6ee2713c6b0ef40e2233ea9208d31b6ce63a5159b66893e8","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-20T10:22:36Z","title_canon_sha256":"1320b7f2fe5f3018f989a3f17c0d456b40274d1ce2dfa24f55ff24310e3b9c67"},"schema_version":"1.0","source":{"id":"2507.14897","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.14897","created_at":"2026-07-05T11:40:15Z"},{"alias_kind":"arxiv_version","alias_value":"2507.14897v1","created_at":"2026-07-05T11:40:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.14897","created_at":"2026-07-05T11:40:15Z"},{"alias_kind":"pith_short_12","alias_value":"QR556MGWIBCL","created_at":"2026-07-05T11:40:15Z"},{"alias_kind":"pith_short_16","alias_value":"QR556MGWIBCLG3HC","created_at":"2026-07-05T11:40:15Z"},{"alias_kind":"pith_short_8","alias_value":"QR556MGW","created_at":"2026-07-05T11:40:15Z"}],"graph_snapshots":[{"event_id":"sha256:689fe3a749ad717a43274ea5d776daafb2b6d387f3bfa1e06e03ae87cd5bca4f","target":"graph","created_at":"2026-07-05T11:40:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.14897/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language model (LM) agents have gained significant attention for their ability to autonomously complete tasks through interactions with environments, tools, and APIs. LM agents are primarily built with prompt engineering or supervised finetuning. At the same time, reinforcement learning (RL) has been explored to enhance LM's capabilities, such as reasoning and factuality. However, the combination of the LM agents and reinforcement learning (Agent-RL) remains underexplored and lacks systematic study. To this end, we built AgentFly, a scalable and extensible Agent-RL framework designed to empowe","authors_text":"Bilal El Bouardi, Fajri Koto, Haonan Li, Renxi Wang, Rifo Ahmad Genadi, Timothy Baldwin, Yongxin Wang, Zhengzhong Liu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-20T10:22:36Z","title":"AgentFly: Extensible and Scalable Reinforcement Learning for LM Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.14897","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8a6a2adfa36274a6c7ea33029fec601ee9f985f302f7a895b87c1877280c799f","target":"record","created_at":"2026-07-05T11:40:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"512f14c2915de54d6ee2713c6b0ef40e2233ea9208d31b6ce63a5159b66893e8","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-20T10:22:36Z","title_canon_sha256":"1320b7f2fe5f3018f989a3f17c0d456b40274d1ce2dfa24f55ff24310e3b9c67"},"schema_version":"1.0","source":{"id":"2507.14897","kind":"arxiv","version":1}},"canonical_sha256":"847bdf30d64044b36ce2fbe7319380ce87f04edf7b3b42f459c74f17641a282e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"847bdf30d64044b36ce2fbe7319380ce87f04edf7b3b42f459c74f17641a282e","first_computed_at":"2026-07-05T11:40:15.248872Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:40:15.248872Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"DgmKtPznRYfOqkmBAPFSWteakrAdHrwBH5MrSJsP+tkHSIXT3Yu4ZvGWyYFC+ewBkk9/01PiBwaWY6qG41x5Cw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:40:15.249296Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.14897","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8a6a2adfa36274a6c7ea33029fec601ee9f985f302f7a895b87c1877280c799f","sha256:689fe3a749ad717a43274ea5d776daafb2b6d387f3bfa1e06e03ae87cd5bca4f"],"state_sha256":"6c1c8433f83cab43ae1d081734a7820c16528132aa19ece1ac4de6c317ffaa45"}