{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:4YRFZKTUV54I66PZLBT7S42D5T","short_pith_number":"pith:4YRFZKTU","canonical_record":{"source":{"id":"2507.19766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-26T03:42:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2658b905f889853c2ae1144a8014d3486e03c0fdf8f54c71ae10cbbff85ba505","abstract_canon_sha256":"3d747a62b2f11fc460fc093a1503bdd5224688c1977e434dcf55e41d1712d353"},"schema_version":"1.0"},"canonical_sha256":"e6225caa74af788f79f95867f97343ecdf873dae01e94629cf545caa8afa6fde","source":{"kind":"arxiv","id":"2507.19766","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.19766","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"arxiv_version","alias_value":"2507.19766v1","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.19766","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"pith_short_12","alias_value":"4YRFZKTUV54I","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"pith_short_16","alias_value":"4YRFZKTUV54I66PZ","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"pith_short_8","alias_value":"4YRFZKTU","created_at":"2026-07-05T11:43:41Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:4YRFZKTUV54I66PZLBT7S42D5T","target":"record","payload":{"canonical_record":{"source":{"id":"2507.19766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-26T03:42:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2658b905f889853c2ae1144a8014d3486e03c0fdf8f54c71ae10cbbff85ba505","abstract_canon_sha256":"3d747a62b2f11fc460fc093a1503bdd5224688c1977e434dcf55e41d1712d353"},"schema_version":"1.0"},"canonical_sha256":"e6225caa74af788f79f95867f97343ecdf873dae01e94629cf545caa8afa6fde","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:41.478350Z","signature_b64":"a13jOk5JBlZkvOfTNsd2Sbkiq+/rLTztbwesqG4jKb8/h/e8yA/lOr28eije5lToYNo94O8VkOpw606mmxl4CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6225caa74af788f79f95867f97343ecdf873dae01e94629cf545caa8afa6fde","last_reissued_at":"2026-07-05T11:43:41.477973Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:41.477973Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2507.19766","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:43:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3FKqn6U4GgLZ4ea3O8A/s9qHIFnTI7gN56g3tB6xwCI8QO0OBSO87ct9JQOTr5HsGKppeb7S9FpIQ2a3m6gaDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T15:27:04.297502Z"},"content_sha256":"10944d4ffdba09cb03a4e2ac1233b76105d46e1c6fcd061eb8534cd87d5c4e3b","schema_version":"1.0","event_id":"sha256:10944d4ffdba09cb03a4e2ac1233b76105d46e1c6fcd061eb8534cd87d5c4e3b"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:4YRFZKTUV54I66PZLBT7S42D5T","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"UloRL:An Ultra-Long Output Reinforcement Learning Approach for Advancing Large Language Models' Reasoning Abilities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dong Du, Shaohua Chen, Shulin Liu, Tao Yang, Yang Li","submitted_at":"2025-07-26T03:42:33Z","abstract_excerpt":"Recent advances in large language models (LLMs) have highlighted the potential of reinforcement learning with verifiable rewards (RLVR) to enhance reasoning capabilities through extended output sequences. However, traditional RL frameworks face inefficiencies when handling ultra-long outputs due to long-tail sequence distributions and entropy collapse during training. To address these challenges, we propose an Ultra-Long Output Reinforcement Learning (UloRL) approach for advancing large language models' reasoning abilities. Specifically, we divide ultra long output decoding into short segments"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.19766","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.19766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:43:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1iuDcYqvi1y2u72o1P3M40hYAkDplgFOLLpIyuCwqk4MiSGjdePdlsG/N5vs+2MeBtiKC+kC7mc5+hrv3vhCAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T15:27:04.298041Z"},"content_sha256":"a51a87b6cc8f954aceff515a42d6995be7f1c3e9ca640dc76b062ef3f2849afb","schema_version":"1.0","event_id":"sha256:a51a87b6cc8f954aceff515a42d6995be7f1c3e9ca640dc76b062ef3f2849afb"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4YRFZKTUV54I66PZLBT7S42D5T/bundle.json","state_url":"https://pith.science/pith/4YRFZKTUV54I66PZLBT7S42D5T/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4YRFZKTUV54I66PZLBT7S42D5T/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T15:27:04Z","links":{"resolver":"https://pith.science/pith/4YRFZKTUV54I66PZLBT7S42D5T","bundle":"https://pith.science/pith/4YRFZKTUV54I66PZLBT7S42D5T/bundle.json","state":"https://pith.science/pith/4YRFZKTUV54I66PZLBT7S42D5T/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4YRFZKTUV54I66PZLBT7S42D5T/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:4YRFZKTUV54I66PZLBT7S42D5T","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3d747a62b2f11fc460fc093a1503bdd5224688c1977e434dcf55e41d1712d353","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-26T03:42:33Z","title_canon_sha256":"2658b905f889853c2ae1144a8014d3486e03c0fdf8f54c71ae10cbbff85ba505"},"schema_version":"1.0","source":{"id":"2507.19766","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.19766","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"arxiv_version","alias_value":"2507.19766v1","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.19766","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"pith_short_12","alias_value":"4YRFZKTUV54I","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"pith_short_16","alias_value":"4YRFZKTUV54I66PZ","created_at":"2026-07-05T11:43:41Z"},{"alias_kind":"pith_short_8","alias_value":"4YRFZKTU","created_at":"2026-07-05T11:43:41Z"}],"graph_snapshots":[{"event_id":"sha256:a51a87b6cc8f954aceff515a42d6995be7f1c3e9ca640dc76b062ef3f2849afb","target":"graph","created_at":"2026-07-05T11:43:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.19766/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advances in large language models (LLMs) have highlighted the potential of reinforcement learning with verifiable rewards (RLVR) to enhance reasoning capabilities through extended output sequences. However, traditional RL frameworks face inefficiencies when handling ultra-long outputs due to long-tail sequence distributions and entropy collapse during training. To address these challenges, we propose an Ultra-Long Output Reinforcement Learning (UloRL) approach for advancing large language models' reasoning abilities. Specifically, we divide ultra long output decoding into short segments","authors_text":"Dong Du, Shaohua Chen, Shulin Liu, Tao Yang, Yang Li","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-26T03:42:33Z","title":"UloRL:An Ultra-Long Output Reinforcement Learning Approach for Advancing Large Language Models' Reasoning Abilities"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.19766","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:10944d4ffdba09cb03a4e2ac1233b76105d46e1c6fcd061eb8534cd87d5c4e3b","target":"record","created_at":"2026-07-05T11:43:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3d747a62b2f11fc460fc093a1503bdd5224688c1977e434dcf55e41d1712d353","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-26T03:42:33Z","title_canon_sha256":"2658b905f889853c2ae1144a8014d3486e03c0fdf8f54c71ae10cbbff85ba505"},"schema_version":"1.0","source":{"id":"2507.19766","kind":"arxiv","version":1}},"canonical_sha256":"e6225caa74af788f79f95867f97343ecdf873dae01e94629cf545caa8afa6fde","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e6225caa74af788f79f95867f97343ecdf873dae01e94629cf545caa8afa6fde","first_computed_at":"2026-07-05T11:43:41.477973Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:43:41.477973Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"a13jOk5JBlZkvOfTNsd2Sbkiq+/rLTztbwesqG4jKb8/h/e8yA/lOr28eije5lToYNo94O8VkOpw606mmxl4CQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:43:41.478350Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.19766","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:10944d4ffdba09cb03a4e2ac1233b76105d46e1c6fcd061eb8534cd87d5c4e3b","sha256:a51a87b6cc8f954aceff515a42d6995be7f1c3e9ca640dc76b062ef3f2849afb"],"state_sha256":"da156a56573ed65d21f6bc686ad48f0d24b37d5e459afb80c9217a53a44072de"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Dxat3Xv9/q4pC4Nx4Krdp+NNBRRHplQMlZXWn7Pvpx+tHsNQXiEtxRIy0u5uSxikjgnkBkEGedEGWrGPXcqRDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T15:27:04.303510Z","bundle_sha256":"7575c29689cc2318cb0bf0fe35dfa7c681864593201849260ad8ccb110bf81bb"}}