{"as_of":"2026-08-05T07:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7137643d1c70e26cf1ca768e5a1d099df39a1d92aefa0f5b3529728012a986a3","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T05:46:44.417762Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2506.13351","last_updated":"2026-05-07T20:19:13Z","snapshot_observed_at":"2026-07-31T17:48:52.945760Z","submitted_at":"2025-06-16T10:43:38Z","title":"Direct Reasoning Optimization: Token-Level Reasoning Reflectivity Meets Rubric Gates for Unverifiable Tasks","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-19T09:48:56.990745Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2506.13351"},"observation_digest":"sha256:80d2e6fc95edfcbbc126b79cb7cc257fcf3ff037c5b6e16d09a504b7a3cde027","observation_id":"7745fa08-d56a-4cc0-aa9c-f06996f5d17d","resolution":{"observed_at":"2026-05-19T09:52:14.093197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2507.17746","last_updated":"2025-10-03T01:55:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-23T17:57:55Z","title":"Rubrics as Rewards: Reinforcement Learning Beyond Verifiable Domains","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T06:07:56.678339Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2507.17746"},"observation_digest":"sha256:ccabd467153efa0defdc11f48093668df1488c23a53afebfe8d7ef76058a1b08","observation_id":"0b6983d6-f37e-4f3d-acf4-257c3df171fc","resolution":{"observed_at":"2026-05-13T06:07:56.851371Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T05:46:44.417762Z","title":"arXiv:2503.23829 [cs]","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.05394","last_updated":"2025-09-05T11:13:40Z","snapshot_observed_at":"2026-08-05T05:46:39.509260Z","submitted_at":"2025-09-05T11:13:40Z","title":"Reverse Browser: Vector-Image-to-Code Generator","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-05T05:46:44.417762Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2509.05394"},"observation_digest":"sha256:f8d50d625383c06afe86664a32ee04cd7a957b053e81fe3974bce4c263fbe7c1","observation_id":"459fb54b-206d-4965-8a55-b0e54227814b","resolution":{"observed_at":"2026-08-05T05:46:44.417762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-04T17:57:52.481344Z","title":"Crossing the reward bridge: Expanding rl with verifiable rewards across diverse domains","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.14257","last_updated":"2026-07-23T12:32:53Z","snapshot_observed_at":"2026-08-04T17:57:40.294406Z","submitted_at":"2025-09-12T15:34:07Z","title":"Student-Centered Distillation Narrows the Agentic Gap Between Small and Large LLMs","version":3},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-04T17:57:52.481344Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2509.14257"},"observation_digest":"sha256:81e2631e6a72dd59caa6a0815a883481c6935acfd7b9f7b9a69f8e225ba4d93d","observation_id":"500c22dd-a511-42f3-b0b0-879be3b0da6b","resolution":{"observed_at":"2026-08-04T17:57:52.481344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2509.25454","last_updated":"2026-04-06T19:16:24Z","snapshot_observed_at":"2026-07-06T22:31:10.099674Z","submitted_at":"2025-09-29T20:00:29Z","title":"DeepSearch: Overcome the Bottleneck of Reinforcement Learning with Verifiable Rewards via Monte Carlo Tree Search","version":4},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-18T12:12:25.437344Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2509.25454"},"observation_digest":"sha256:aa5598dd534d6bc7fdfb7ae21a21a36e852b2a139a051e5121c97cd0c5cbe14a","observation_id":"21974947-640f-41f2-956a-4a4e42720073","resolution":{"observed_at":"2026-05-18T12:12:35.834905Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-04T10:40:48.485814Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.09278","last_updated":"2026-06-28T16:27:56Z","snapshot_observed_at":"2026-08-04T10:40:40.009013Z","submitted_at":"2025-10-10T11:21:09Z","title":"CLARity: Reasoning Consistency Alone Can Teach Reinforced Experts","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T10:40:48.485814Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2510.09278"},"observation_digest":"sha256:c47e828a65fe88960506d222f53a24609b20fa54865f1f6c64286c4c95417a7b","observation_id":"5678ad83-14b1-4b82-b552-9a4d16097830","resolution":{"observed_at":"2026-08-04T10:40:48.485814Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2510.10649","last_updated":"2026-04-16T14:51:36Z","snapshot_observed_at":"2026-07-06T22:32:27.698678Z","submitted_at":"2025-10-12T15:06:53Z","title":"Unlocking Exploration in RLVR: Uncertainty-aware Advantage Shaping for Deeper Reasoning","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-18T07:20:01.505216Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2510.10649"},"observation_digest":"sha256:610a84f1d0e1eff9159adc88a8a3f43690757c2fde17de260c93a4672ce6a9e5","observation_id":"f6d6e0de-6812-4916-84ef-98aeff15b891","resolution":{"observed_at":"2026-05-18T07:21:04.360338Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2512.03847","last_updated":"2026-05-06T14:15:19Z","snapshot_observed_at":"2026-08-03T04:49:11.537109Z","submitted_at":"2025-12-03T14:48:38Z","title":"DVPO: Distributional Value Modeling-based Policy Optimization for LLM Post-Training","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T01:46:21.744857Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2512.03847"},"observation_digest":"sha256:53b1f896d315fe561be327abf0226cc314a7454d1358407a1c2f075c2b33fac7","observation_id":"49959a33-24e4-4b0d-861a-15ade233c683","resolution":{"observed_at":"2026-05-17T01:48:50.980608Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2601.13262","last_updated":"2026-04-26T01:38:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-19T17:51:00Z","title":"CURE-Med: Curriculum-Informed Reinforcement Learning for Multilingual Medical Reasoning","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T13:20:49.919833Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2601.13262"},"observation_digest":"sha256:839f39f93949029d6fd9ebbef07472331bdb4eaba0ae5eaab068414f1d2d64b1","observation_id":"553adc0d-e76a-4293-a47a-83b6d66855ed","resolution":{"observed_at":"2026-05-16T13:20:57.559827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2603.03197","last_updated":"2026-04-12T16:48:44Z","snapshot_observed_at":"2026-08-03T05:43:55.769983Z","submitted_at":"2026-03-03T17:52:39Z","title":"Specificity-aware reinforcement learning for fine-grained open-world classification","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-15T16:44:47.866126Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2603.03197"},"observation_digest":"sha256:ef83fef3f90c8921feba9a21fd39749c69d4e1dc365b9d0035607ad2ccf4fdd5","observation_id":"53947a8e-5330-48d8-88c6-f47d8edd3061","resolution":{"observed_at":"2026-05-15T16:46:17.815819Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2603.16876","last_updated":"2026-05-08T08:14:14Z","snapshot_observed_at":"2026-08-02T05:44:39.173956Z","submitted_at":"2026-02-17T12:48:32Z","title":"Multi-Modal Multi-Agent Reinforcement Learning for Radiology Report Generation","version":2},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-05-15T21:51:10.972744Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2603.16876"},"observation_digest":"sha256:2b9ed4cef03aa0733f29f7cae7de01e560986d29ea16fe9006a82464a4418b5e","observation_id":"1570ade8-c9af-4dc6-bf9f-63b133f018d8","resolution":{"observed_at":"2026-05-15T21:51:40.644170Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-07-13T20:08:17.066898Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.22744","last_updated":"2026-05-29T00:35:02Z","snapshot_observed_at":"2026-08-03T18:04:48.244219Z","submitted_at":"2026-03-24T03:16:32Z","title":"LH-Bench: Skill-Grounded Evaluation of Long-Horizon Agents on Subjective Enterprise Tasks","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-13T20:08:17.066898Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2603.22744"},"observation_digest":"sha256:f781a3c36a8a69fad8bf3f412f40fb2ed99e96c15cae851b709a705b1c80ebdf","observation_id":"3b7189d9-5807-4d9e-aa2f-658142edfee5","resolution":{"observed_at":"2026-07-13T20:08:17.066898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-07-13T14:23:49.346727Z","title":"Crossing the reward bridge: Expanding rl with verifiable rewards across diverse domains","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.01475","last_updated":"2026-06-08T13:23:37Z","snapshot_observed_at":"2026-07-13T14:23:49.176171Z","submitted_at":"2026-04-01T23:31:38Z","title":"Interpretable Electrophysiological Features of Resting-State EEG Capture Cortical Network Dynamics in Parkinsons Disease","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-13T14:23:49.346727Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2604.01475"},"observation_digest":"sha256:b5e6a23cc2ab536d00cb0388d220415ae16ea45cfd8a6ca03b5fa0ba5d15140d","observation_id":"a62ac229-1b51-47de-b77c-2b9ccaf8fd5b","resolution":{"observed_at":"2026-07-13T14:23:49.346727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2604.10110","last_updated":"2026-04-11T09:08:27Z","snapshot_observed_at":"2026-07-06T22:58:47.599383Z","submitted_at":"2026-04-11T09:08:27Z","title":"Trust Your Memory: Verifiable Control of Smart Homes through Reinforcement Learning with Multi-dimensional Rewards","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:49:00.343580Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2604.10110"},"observation_digest":"sha256:88de1bbcd598ee56219a7543a05136a167c177d42c899bf5357ce5739add8f0c","observation_id":"cbb70e96-f794-4e69-a379-9ec69c09355e","resolution":{"observed_at":"2026-05-11T08:06:01.172294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2604.20755","last_updated":"2026-04-22T16:44:33Z","snapshot_observed_at":"2026-08-01T03:59:28.787636Z","submitted_at":"2026-04-22T16:44:33Z","title":"V-tableR1: Process-Supervised Multimodal Table Reasoning with Critic-Guided Policy Optimization","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-09T23:48:32.613988Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2604.20755"},"observation_digest":"sha256:4757834a9d51199c194102a1d0ccd4c06bb4d08501b5e1654f3990c79413c357","observation_id":"369ad831-a096-4da0-8abc-13cba5e99796","resolution":{"observed_at":"2026-05-11T14:01:04.148354Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2605.00754","last_updated":"2026-05-08T13:29:37Z","snapshot_observed_at":"2026-07-06T23:14:06.327027Z","submitted_at":"2026-05-01T16:07:34Z","title":"Themis: Training Robust Multilingual Code Reward Models for Flexible Multi-Criteria Scoring","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-09T19:33:35.690030Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2605.00754"},"observation_digest":"sha256:83dea92fcc21243adebb01c20fa808eaab326c5351aca4ebdebe80e06d6ba63c","observation_id":"5803ac88-fbb8-4ee0-bba0-c31b7b85402d","resolution":{"observed_at":"2026-05-09T19:35:39.078929Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2605.00754","last_updated":"2026-05-08T13:29:37Z","snapshot_observed_at":"2026-07-06T23:14:06.327027Z","submitted_at":"2026-05-01T16:07:34Z","title":"Themis: Training Robust Multilingual Code Reward Models for Flexible Multi-Criteria Scoring","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T01:56:43.707252Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2605.00754"},"observation_digest":"sha256:61fd5fdc584425bc7b04135e3434af6319ca22db3e72a033c5d7d6dc39bbde9f","observation_id":"b3786b5d-2427-4e79-8e10-970e17cfd6c1","resolution":{"observed_at":"2026-05-11T02:15:52.434975Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2605.02913","last_updated":"2026-04-08T00:53:29Z","snapshot_observed_at":"2026-07-06T23:15:55.848885Z","submitted_at":"2026-04-08T00:53:29Z","title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","version":1},"reference_index":105,"source":"arxiv_source","source_observed_at":"2026-05-10T19:15:27.406778Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2605.02913"},"observation_digest":"sha256:7289db330d3865f8b43cbd6537982f1a9a1da189522ec4e29bf9bc909a88bfa0","observation_id":"1dd83977-cef3-48d3-8789-8ceb9697ab7a","resolution":{"observed_at":"2026-05-10T23:15:49.250698Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2605.09879","last_updated":"2026-05-11T02:05:30Z","snapshot_observed_at":"2026-07-06T23:21:54.140120Z","submitted_at":"2026-05-11T02:05:30Z","title":"M2A: Synergizing Mathematical and Agentic Reasoning in Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-12T04:43:50.384980Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2605.09879"},"observation_digest":"sha256:b26b7f4e7c3a9c95f8a8c16154261bb8cc640116eab340a33923c2a24df5d055","observation_id":"6e20c37c-f7c7-41eb-ab24-00109f9543f2","resolution":{"observed_at":"2026-05-12T06:01:22.662442Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2605.20061","last_updated":"2026-05-19T16:19:29Z","snapshot_observed_at":"2026-07-31T19:01:19.361800Z","submitted_at":"2026-05-19T16:19:29Z","title":"Rewarding Beliefs, Not Actions: Consistency-Guided Credit Assignment for Long-Horizon Agents","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T05:35:45.084011Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2605.20061"},"observation_digest":"sha256:6c64b8009072bad661c6909a706e4fb210e1772555757933c4e1323fb4e19a8a","observation_id":"c2ca204c-6241-40e0-afc0-1f11f80191cb","resolution":{"observed_at":"2026-05-20T05:38:05.525894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2605.31058","last_updated":"2026-05-29T09:29:32Z","snapshot_observed_at":"2026-08-04T04:33:32.255796Z","submitted_at":"2026-05-29T09:29:32Z","title":"Combinatorial Synthesis: Scaling Code RLVR via Atomic Decomposition and Recombination","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-28T23:08:05.597810Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2605.31058"},"observation_digest":"sha256:4546a6b12285df43f5aa4e600cdd6b88d8e0d10e36750589e69579e8849ab40a","observation_id":"1f31b556-3816-4be0-bc71-8924d6b7bc72","resolution":{"observed_at":"2026-06-28T23:12:47.109855Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.00172","last_updated":"2026-05-29T13:21:30Z","snapshot_observed_at":"2026-08-03T13:58:16.997854Z","submitted_at":"2026-05-29T13:21:30Z","title":"CAST: Non-Privileged Clipped Asymmetric Self-Teaching with Advantage Flipping for GRPO","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T22:25:04.067339Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.00172"},"observation_digest":"sha256:027cfd7dec8e2d16233de15ddada38e3005c1b352181c44eec370eae0e158ec3","observation_id":"bfe80b35-bcae-4429-a91f-e2caf0a75a60","resolution":{"observed_at":"2026-07-01T19:26:01.064488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.00609","last_updated":"2026-05-30T08:18:40Z","snapshot_observed_at":"2026-07-06T23:41:15.410587Z","submitted_at":"2026-05-30T08:18:40Z","title":"CARE-RL: Capability-Aware Reinforcement Learning for Mitigating Cross-Domain Conflicts","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-06-28T19:01:14.340754Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.00609"},"observation_digest":"sha256:c6e528a58875d5cf38fbdd4dc83de7d9a70c4d193c78583bc42417ecc980e25c","observation_id":"f305aa37-c9eb-46a2-9aa9-e80b02291d7d","resolution":{"observed_at":"2026-06-28T19:02:33.947566Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.04516","last_updated":"2026-06-03T06:47:50Z","snapshot_observed_at":"2026-07-06T23:44:37.797802Z","submitted_at":"2026-06-03T06:47:50Z","title":"GeoMin: Data-Efficient Semi-Supervised RLVR via Geometric Distribution Modeling","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-28T07:45:43.320339Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.04516"},"observation_digest":"sha256:918835d1005b11d4f35550b714fe6dc785da980eb8235f56bc0ff69fe3d85c7e","observation_id":"0569465f-718a-4fb8-91cd-7afc15e533d5","resolution":{"observed_at":"2026-07-02T06:06:40.787545Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.09393","last_updated":"2026-06-08T12:09:20Z","snapshot_observed_at":"2026-07-06T23:48:44.159537Z","submitted_at":"2026-06-08T12:09:20Z","title":"CapRL++: Unified Reinforcement Learning with Verifiable Rewards for Dense Image and Video Captioning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-27T17:21:38.543724Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.09393"},"observation_digest":"sha256:d65ccddc48945de0b27ea3980cebace2b6e39e1ad76bed4c9a2237df20890719","observation_id":"69cbd603-7104-40b4-853a-246a243078a3","resolution":{"observed_at":"2026-07-03T00:17:29.019297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.23557","last_updated":"2026-06-22T16:28:11Z","snapshot_observed_at":"2026-08-05T01:24:53.186762Z","submitted_at":"2026-06-22T16:28:11Z","title":"Dense Reward for Multi-View 3D Reasoning with Global Maps and Local Views","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-26T08:38:46.044079Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.23557"},"observation_digest":"sha256:2d74a616a7816661be642f18e7d97677fd88cd26788e5813287987995405b208","observation_id":"55788ff5-3d72-4a1a-888c-f2014b89a2f6","resolution":{"observed_at":"2026-07-04T10:39:45.338242Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.24539","last_updated":"2026-06-23T13:06:51Z","snapshot_observed_at":"2026-08-02T12:19:25.025255Z","submitted_at":"2026-06-23T13:06:51Z","title":"PointVG-R: Internalizing Geometric Reasoning in MLLMs for Precise Pointing Localization via Visual Chain of Thought","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-26T00:46:17.094339Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.24539"},"observation_digest":"sha256:e0a46191ced2c0b72d9b1a8f56029091645d66184dd0a6f8eb9921f2286cbc48","observation_id":"1bb171b7-5dd2-4e15-b5ba-78bbaeb0644f","resolution":{"observed_at":"2026-07-04T16:19:57.794350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2606.28707","last_updated":"2026-06-27T03:25:53Z","snapshot_observed_at":"2026-07-07T00:02:47.924620Z","submitted_at":"2026-06-27T03:25:53Z","title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-06-30T10:07:39.554999Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2606.28707"},"observation_digest":"sha256:c4985caa00644ea5b936a6db330e11e0d17e73ed66027a08a2d86e98ada770a8","observation_id":"dcf008c4-a9b1-4f38-9e35-7bf8114e9a0c","resolution":{"observed_at":"2026-06-30T12:44:40.146628Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2607.00139","last_updated":"2026-06-30T20:18:32Z","snapshot_observed_at":"2026-08-03T00:21:17.843857Z","submitted_at":"2026-06-30T20:18:32Z","title":"Benchmarking Frontier LLMs on Arabic Cultural and Sociolinguistic Knowledge: A Cross-Evaluation Framework with Human SME Ground Truth","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-02T19:26:21.833145Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2607.00139"},"observation_digest":"sha256:532bd586dede4b16077ca0f08703c7a819d058f9e56aa72c333c3ef4f7598a1f","observation_id":"61f6e4fe-1eae-489a-950e-6014c1940867","resolution":{"observed_at":"2026-07-02T19:27:18.482991Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains","version":2},"cited_work":{"arxiv_id":"2503.23829","doi":"10.48550/arxiv.2503.23829","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23829","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Crossing the reward bridge: Expanding RL with verifiable rewards across diverse domains","venue":"ArXiv.org","work_id":"a1dbfba1-b433-4260-abc6-bf28628844b9","year":2025},"citing_paper":{"arxiv_id":"2607.02407","last_updated":"2026-07-02T16:40:08Z","snapshot_observed_at":"2026-08-02T06:58:46.648135Z","submitted_at":"2026-07-02T16:40:08Z","title":"Text-Driven 3D Indoor Scene Synthesis in Non-Manhattan Environments","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-07-03T13:21:06.380698Z"},"links":{"cited_paper":"/paper/2503.23829","citing_paper":"/paper/2607.02407"},"observation_digest":"sha256:87ccd0a847bbf4de7496d4301d2345ab4d0628246fd7ccd1da7bd42540b20cd4","observation_id":"0c369fed-3427-4cc4-84b1-b2f2437d9bc2","resolution":{"observed_at":"2026-07-03T13:28:18.294461Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2503.23829/citation-record","integrity":"/paper/2503.23829/integrity","json":"/paper/2503.23829/citation-record.json","paper":"/paper/2503.23829"},"outbound":[],"paper":{"arxiv_id":"2503.23829","last_updated":"2025-04-01T14:48:02Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T21:01:17.931818Z","submitted_at":"2025-03-31T08:22:49Z","title":"Crossing the Reward Bridge: Expanding RL with Verifiable Rewards Across Diverse Domains"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2503.23829."}