{"as_of":"2026-08-06T06:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4ea852c856bb16cbbc21040528ef755a5bcd3a6e7a8ee759bd352a8b152fc694","coverage":[{"denominator":41,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":41,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T00:13:30.397515Z","state":"measured"},{"denominator":41,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":41,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2602.11351/citation-record","integrity":"/paper/2602.11351/integrity","json":"/paper/2602.11351/citation-record.json","paper":"/paper/2602.11351"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:24.996133Z","title":"Consistently simulating human personas with multi-turn reinforcement learning.arXiv preprint arXiv:2511.00222,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:24.996133Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:93f46d089bd5615e119d90c54c5ffb5bb2b961961d217e496f980f468712fb1e","observation_id":"4cf34756-4688-4a94-a3f3-eb8df2a296f7","resolution":{"observed_at":"2026-08-03T00:13:24.996133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.01600","last_updated":"2025-03-08T05:23:57Z","snapshot_observed_at":"2026-07-06T20:30:26.881629Z","submitted_at":"2025-02-03T18:35:42Z","title":"Reinforcement Learning for Long-Horizon Interactive LLM Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.01600","snapshot_observed_at":"2026-08-03T00:13:25.232880Z","title":"Reinforce- ment learning for long-horizon interactive llm agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.232880Z"},"links":{"cited_paper":"/paper/2502.01600","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:ebd0d4c361b67e175faa01f898f61274cdb5a849d387900ce277b89480f850e2","observation_id":"925a5ce1-d2d7-4178-bfe5-2fb732e95e7c","resolution":{"observed_at":"2026-08-03T00:13:25.232880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01307","last_updated":"2025-08-15T15:21:46Z","snapshot_observed_at":"2026-07-06T20:45:32.493589Z","submitted_at":"2025-03-03T08:46:22Z","title":"Cognitive Behaviors that Enable Self-Improving Reasoners, or, Four Habits of Highly Effective STaRs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01307","snapshot_observed_at":"2026-08-03T00:13:25.386643Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.386643Z"},"links":{"cited_paper":"/paper/2503.01307","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:3c791b8ffdb958081586d2efb60d931c88957787dbdaf7f24a2ec53d8127ed36","observation_id":"5d47239c-cc22-4a5b-98c7-0fd887ce84b7","resolution":{"observed_at":"2026-08-03T00:13:25.386643Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-03T00:13:25.542401Z","title":"Deepseek-r1: In- centivizing reasoning capability in llms via reinforcement learning.arXiv preprint arXiv:2501.12948,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.542401Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:0eeb0ed7f4bc9c47e51f99f95ecf6a22112dd1f4d31ac50a9131946a2bdc0a7b","observation_id":"2f6680bd-5aac-44f1-a237-74399e3fa718","resolution":{"observed_at":"2026-08-03T00:13:25.542401Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1502.02259","last_updated":"2015-02-08T14:58:50Z","snapshot_observed_at":"2026-08-03T17:17:59.241164Z","submitted_at":"2015-02-08T14:58:50Z","title":"Contextual Markov Decision Processes","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1502.02259","snapshot_observed_at":"2026-08-03T00:13:25.745024Z","title":"Con- textual markov decision processes.arXiv preprint arXiv:1502.02259,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.745024Z"},"links":{"cited_paper":"/paper/1502.02259","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:31106d29d246fb7f36eabd473162e8a66a0fe414cf5af739167e3190f3808822","observation_id":"de66711d-3d1d-4196-a7cb-7a50aa09c23d","resolution":{"observed_at":"2026-08-03T00:13:25.745024Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:26.242813Z","title":"R., He, J., Yu, H., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.242813Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:de1c3a7b1ebd86d0b4879ff08b2c21e46ef725c28c5972f810f7f3846c16c5a8","observation_id":"d3cd077b-4b50-4363-b7c5-5ba09127493d","resolution":{"observed_at":"2026-08-03T00:13:26.242813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:26.329894Z","title":"Quagmires in sft-rl post- training: When high sft scores mislead and what to use instead.arXiv preprint arXiv:2510.01624,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.329894Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:8f7b92c02a90f02c4cfb3e681b5c5964b6a3e359936cf65700989cebc17ccd62","observation_id":"ed6abef3-9eaa-4136-af20-fc317702da7d","resolution":{"observed_at":"2026-08-03T00:13:26.329894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01679","last_updated":"2025-06-03T20:51:06Z","snapshot_observed_at":"2026-08-05T20:06:13.107818Z","submitted_at":"2024-10-02T15:49:30Z","title":"VinePPO: Refining Credit Assignment in RL Training of LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01679","snapshot_observed_at":"2026-08-03T00:13:26.456107Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.456107Z"},"links":{"cited_paper":"/paper/2410.01679","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:867018b5c8934da5a0f15f27419aa881bf2a4ca5819a188c8e0d25c40b0175ef","observation_id":"fdafd0cb-8359-4589-80fa-a54d0b1c4e9f","resolution":{"observed_at":"2026-08-03T00:13:26.456107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17428","last_updated":"2025-02-25T00:35:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-27T17:59:45Z","title":"NV-Embed: Improved Techniques for Training LLMs as Generalist Embedding Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.17428","snapshot_observed_at":"2026-08-03T00:13:26.575467Z","title":"Nv-embed: Improved techniques for training llms as generalist embedding models.arXiv preprint arXiv:2405.17428,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.575467Z"},"links":{"cited_paper":"/paper/2405.17428","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:fb967a21b8411f4162bde8b076d700bf6d3c43bb8170a10224dba7fe5aa19eae","observation_id":"1dbddb93-2803-4ca7-88e5-63fe4e100703","resolution":{"observed_at":"2026-08-03T00:13:26.575467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14106","last_updated":"2025-05-25T18:28:20Z","snapshot_observed_at":"2026-07-06T21:26:54.693576Z","submitted_at":"2025-05-20T09:13:22Z","title":"A Personalized Conversational Benchmark: Towards Simulating Personalized Conversations","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14106","snapshot_observed_at":"2026-08-03T00:13:26.701496Z","title":"A., Dernoncourt, F., Kveton, B., Wu, J., Yu, T., Song, L., Yang, T., Qin, Y ., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.701496Z"},"links":{"cited_paper":"/paper/2505.14106","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:fb0d74fb707cb6a9ae77dc662a9b8add57dc25fa625e9e7d8c141742a96c6226","observation_id":"bd3124c4-1797-4d42-9b63-7bc0125cb811","resolution":{"observed_at":"2026-08-03T00:13:26.701496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.14295","last_updated":"2025-08-22T16:49:10Z","snapshot_observed_at":"2026-08-02T08:06:42.944434Z","submitted_at":"2025-07-18T18:07:38Z","title":"A Simple \"Try Again\" Can Elicit Multi-Turn LLM Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.14295","snapshot_observed_at":"2026-08-03T00:13:26.854490Z","title":"Let’s try again: Eliciting multi-turn reasoning 9 Behavioral Agentic Optimization in language models via simplistic feedback","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.854490Z"},"links":{"cited_paper":"/paper/2507.14295","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:30efd4ee6ddd66954fe38b980770057137601884b906228994f538db51851849","observation_id":"f75ca2ed-f3e2-4825-8c23-754567efa4a3","resolution":{"observed_at":"2026-08-03T00:13:26.854490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21620","last_updated":"2025-05-24T08:46:08Z","snapshot_observed_at":"2026-08-01T20:06:59.931737Z","submitted_at":"2025-03-27T15:39:30Z","title":"UI-R1: Enhancing Efficient Action Prediction of GUI Agents by Reinforcement Learning","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21620","snapshot_observed_at":"2026-08-03T00:13:27.015676Z","title":"Ui-r1: Enhancing efficient action prediction of gui agents by reinforcement learning.arXiv preprint arXiv:2503.21620,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.015676Z"},"links":{"cited_paper":"/paper/2503.21620","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:98a0701144124ecd1e8f9d162425f8d9769483130a60192ad4a6a74832348cf3","observation_id":"21967e0c-e5f6-42a9-a2a2-d93f2d683900","resolution":{"observed_at":"2026-08-03T00:13:27.015676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10458","last_updated":"2025-10-01T04:55:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:45:54Z","title":"GUI-R1 : A Generalist R1-Style Vision-Language Action Model For GUI Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10458","snapshot_observed_at":"2026-08-03T00:13:27.189438Z","title":"Gui- r1: A generalist r1-style vision-language action model for gui agents.arXiv preprint arXiv:2504.10458,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.189438Z"},"links":{"cited_paper":"/paper/2504.10458","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:4e465d04afa64cb1fd9b2286a8710384035613620caee4bab5b7c07b4a234660","observation_id":"c097a627-c7e7-408b-87bf-6b323c113e63","resolution":{"observed_at":"2026-08-03T00:13:27.189438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.07152","last_updated":"2025-09-07T02:40:21Z","snapshot_observed_at":"2026-08-05T05:33:12.830903Z","submitted_at":"2024-11-11T17:28:19Z","title":"HierTOD: A Task-Oriented Dialogue System Driven by Hierarchical Goals","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.07152","snapshot_observed_at":"2026-08-03T00:13:27.347739Z","title":"Hiertod: A task-oriented dialogue system driven by hier- archical goals.arXiv preprint arXiv:2411.07152,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.347739Z"},"links":{"cited_paper":"/paper/2411.07152","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:f7feed6c6e351a2d78a3dd0276feb7a534a0ab0a742b8547b0385cd59792c82f","observation_id":"ab94e218-f79c-409f-8062-ecaef6c54f67","resolution":{"observed_at":"2026-08-03T00:13:27.347739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11827","last_updated":"2025-05-25T04:09:29Z","snapshot_observed_at":"2026-08-05T09:49:18.872955Z","submitted_at":"2025-05-17T04:26:39Z","title":"Not All Thoughts are Generated Equal: Efficient LLM Reasoning via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11827","snapshot_observed_at":"2026-08-03T00:13:27.574670Z","title":"Not all thoughts are generated equal: Efficient llm reasoning via multi-turn reinforcement learning.arXiv preprint arXiv:2505.11827,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.574670Z"},"links":{"cited_paper":"/paper/2505.11827","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:16497bcad2a83a9606b66619b386963683473b8541d2adbc7acc88b1e31746f2","observation_id":"a8a8e0be-0c63-4936-a4fb-ca7f1444b804","resolution":{"observed_at":"2026-08-03T00:13:27.574670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.13543","last_updated":"2025-04-01T14:45:22Z","snapshot_observed_at":"2026-07-06T19:53:21.444233Z","submitted_at":"2024-11-20T18:54:32Z","title":"BALROG: Benchmarking Agentic LLM and VLM Reasoning On Games","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.13543","snapshot_observed_at":"2026-08-03T00:13:27.793264Z","title":"Balrog: Benchmarking agentic llm and vlm reasoning on games.arXiv preprint arXiv:2411.13543,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.793264Z"},"links":{"cited_paper":"/paper/2411.13543","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:37bfad88fd69ab3b1841f521922c5169f072ceb6d0ee4e81fb3ae968a8a6425a","observation_id":"97c2fc39-6466-4cf3-a527-32042fbe3344","resolution":{"observed_at":"2026-08-03T00:13:27.793264Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.03601","last_updated":"2025-07-19T17:39:17Z","snapshot_observed_at":"2026-08-05T07:23:13.534267Z","submitted_at":"2025-04-04T17:13:57Z","title":"APIGen-MT: Agentic Pipeline for Multi-Turn Data Generation via Simulated Agent-Human Interplay","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.03601","snapshot_observed_at":"2026-08-03T00:13:27.863872Z","title":"C., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.863872Z"},"links":{"cited_paper":"/paper/2504.03601","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:2f3b9542423e442bee72d54ae0fcdd6d08bfbc28aab0311ae32829073705046c","observation_id":"670737dc-3c3a-42ab-a041-0bf224cc8170","resolution":{"observed_at":"2026-08-03T00:13:27.863872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13958","last_updated":"2025-04-16T21:45:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-16T21:45:32Z","title":"ToolRL: Reward is All Tool Learning Needs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.13958","snapshot_observed_at":"2026-08-03T00:13:27.985409Z","title":"C., He, Q., Wang, H., Chen, X., Hakkani-T¨ur, D., Tur, G., and Ji, H","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.985409Z"},"links":{"cited_paper":"/paper/2504.13958","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:872c37ed808e4afa44b11f04e46d88fb40659f7bcbcfa4cdfba764cee12afed6","observation_id":"1f86f967-51b6-442c-a056-e36475a4f012","resolution":{"observed_at":"2026-08-03T00:13:27.985409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-03T00:13:28.112926Z","title":"Deepseekmath: Push- ing the limits of mathematical reasoning in open language models.arXiv preprint arXiv:2402.03300,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.112926Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:5ef5c5b23bea71e059d846f8f1abbf01a3d92323999df4a6b52cb2d0945fe27c","observation_id":"df72b7ae-2db8-4bda-985b-47de9f1a2166","resolution":{"observed_at":"2026-08-03T00:13:28.112926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02508","last_updated":"2025-06-16T03:29:47Z","snapshot_observed_at":"2026-08-05T03:22:49.530240Z","submitted_at":"2025-02-04T17:26:58Z","title":"Satori: Reinforcement Learning with Chain-of-Action-Thought Enhances LLM Reasoning via Autoregressive Search","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02508","snapshot_observed_at":"2026-08-03T00:13:28.211988Z","title":"Satori: Reinforcement learning with chain-of-action-thought en- hances llm reasoning via autoregressive search.arXiv preprint arXiv:2502.02508,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.211988Z"},"links":{"cited_paper":"/paper/2502.02508","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:d29a50d90e49939136cfc10309d922f06e21187bfcbd3ae808ef5b6b8ebaf750","observation_id":"2465e9b8-0ca7-4897-9838-026eacf3cd3d","resolution":{"observed_at":"2026-08-03T00:13:28.211988Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06358","last_updated":"2025-03-08T23:41:20Z","snapshot_observed_at":"2026-07-06T20:49:13.767335Z","submitted_at":"2025-03-08T23:41:20Z","title":"Language Model Personalization via Reward Factorization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06358","snapshot_observed_at":"2026-08-03T00:13:28.352379Z","title":"Language model personalization via reward factorization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.352379Z"},"links":{"cited_paper":"/paper/2503.06358","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:c48e9412b40bcf2b3d57dd103ee6fe91467fe7b973f8921bb58ee5f073ddcf62","observation_id":"fbaf1417-b5da-400e-9a9f-40f7dd5c0742","resolution":{"observed_at":"2026-08-03T00:13:28.352379Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.01441","last_updated":"2025-04-28T10:42:49Z","snapshot_observed_at":"2026-08-05T20:36:54.358831Z","submitted_at":"2025-04-28T10:42:49Z","title":"Agentic Reasoning and Tool Integration for LLMs via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.01441","snapshot_observed_at":"2026-08-03T00:13:28.480694Z","title":"Agentic reasoning and tool integration for llms via reinforcement learning.arXiv preprint arXiv:2505.01441,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.480694Z"},"links":{"cited_paper":"/paper/2505.01441","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:81731a484db01df57ddf1f4f0fc892be3cd08ebb58c9efe2c84b2973e5e46938","observation_id":"fc728a6f-6756-4464-bffe-ece9910977f7","resolution":{"observed_at":"2026-08-03T00:13:28.480694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:28.610154Z","title":"Training proactive and per- sonalized llm agents.arXiv preprint arXiv:2511.02208,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.610154Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:721e67efb1e15798061379f04cfe514a7edabbe7af4e9df52ddafab21b8be4e1","observation_id":"d746fc4d-08f7-4f89-804f-0866a12d889f","resolution":{"observed_at":"2026-08-03T00:13:28.610154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-03T00:13:28.684575Z","title":"M., Hauth, A., Millican, K., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.684575Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:2559fb80688380d03527b7dca7e76d35adddb20d3dbf754ec4273b25c34e3834","observation_id":"0b738638-e145-4a6d-a552-07f26960a010","resolution":{"observed_at":"2026-08-03T00:13:28.684575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:28.984404Z","title":"En- hancing personalized multi-turn dialogue with curiosity reward.arXiv preprint arXiv:2504.03206,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.984404Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:d4c37645909545a547c54a34a6bf09cefc15f2733380c27d498db214fad6b7a0","observation_id":"10028b63-61cd-4504-8a77-e75ff9eb10a6","resolution":{"observed_at":"2026-08-03T00:13:28.984404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.05606","last_updated":"2026-05-17T17:20:46Z","snapshot_observed_at":"2026-07-06T21:37:37.394452Z","submitted_at":"2025-06-05T21:37:49Z","title":"OPeRA: A Dataset of Observation, Persona, Rationale, and Action for Evaluating LLMs on Human Online Shopping Behavior Simulation","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05606","snapshot_observed_at":"2026-08-03T00:13:29.080919Z","title":"Opera: A dataset of observation, persona, rationale, and action for evaluating llms on human online shopping behavior simulation.arXiv preprint arXiv:2506.05606, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.080919Z"},"links":{"cited_paper":"/paper/2506.05606","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:139c924a7e7f4222ce925b659561ab542cc066d60aa9793c753249501db41c25","observation_id":"9391ca7b-429b-4994-a220-a8f2c6f0f8b9","resolution":{"observed_at":"2026-08-03T00:13:29.080919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:29.386885Z","title":"Boad: Discovering hierarchi- cal software engineering agents via bandit optimization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.386885Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:441d26d977103eca5a03b0f2423eea40ccc6d563f88130c43513700d1f1bfbdb","observation_id":"b279e7df-c54c-4fe5-a995-6298419c2e5b","resolution":{"observed_at":"2026-08-03T00:13:29.386885Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-03T00:13:29.423336Z","title":"Qwen3 technical report.arXiv preprint arXiv:2505.09388, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.423336Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:4900304562d8609558571a6d119e47e59504cf18587362882549c03fcb789691","observation_id":"43f0cd01-802e-484c-b23a-15c30fe2faac","resolution":{"observed_at":"2026-08-03T00:13:29.423336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.03373","last_updated":"2025-02-05T17:13:32Z","snapshot_observed_at":"2026-07-06T20:31:41.231839Z","submitted_at":"2025-02-05T17:13:32Z","title":"Demystifying Long Chain-of-Thought Reasoning in LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.03373","snapshot_observed_at":"2026-08-03T00:13:29.528540Z","title":"Demys- tifying long chain-of-thought reasoning in llms.arXiv preprint arXiv:2502.03373,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.528540Z"},"links":{"cited_paper":"/paper/2502.03373","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:676058784ed0af0033de81bfc5c56646068f9d570cdd7f5ddbbc2cc7e49f1227","observation_id":"3bd0f250-4516-44ba-9c1a-1f3650c12246","resolution":{"observed_at":"2026-08-03T00:13:29.528540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:29.699949Z","title":"Demysti- fying reinforcement learning in agentic reasoning.arXiv preprint arXiv:2510.11701,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.699949Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:252d4c629e66b12aa66a79e720c5bb70a9edec7774eaedba2dd2170422a473dd","observation_id":"23557dd4-11bf-41ca-905a-c1c1bfb8f343","resolution":{"observed_at":"2026-08-03T00:13:29.699949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23604","last_updated":"2025-05-29T16:15:36Z","snapshot_observed_at":"2026-08-05T16:25:12.354474Z","submitted_at":"2025-05-29T16:15:36Z","title":"Satori-SWE: Evolutionary Test-Time Scaling for Sample-Efficient Software Engineering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23604","snapshot_observed_at":"2026-08-03T00:13:29.838143Z","title":"Satori- swe: Evolutionary test-time scaling for sample-efficient software engineering.arXiv preprint arXiv:2505.23604, 2025a","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.838143Z"},"links":{"cited_paper":"/paper/2505.23604","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:9bc5cd89d454debdd25548582426dd62b7392bb6209649ef65a085411c36c47d","observation_id":"91f78ddb-4cbf-4659-9b2a-0d815ea1ca68","resolution":{"observed_at":"2026-08-03T00:13:29.838143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:30.021924Z","title":"Teaching language models to evolve with users: Dynamic profile modeling for personalized alignment.arXiv preprint arXiv:2505.15456,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.021924Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:c5b570a2f7b2bfad753eeb4a6a167260a501373a7e917dd1c5a6e9dcdc9a583d","observation_id":"10def8fe-8f92-4e75-af87-e71342188605","resolution":{"observed_at":"2026-08-03T00:13:30.021924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.13372","last_updated":"2024-06-27T22:44:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-20T08:08:54Z","title":"LlamaFactory: Unified Efficient Fine-Tuning of 100+ Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.13372","snapshot_observed_at":"2026-08-03T00:13:30.206953Z","title":"L., Huang, J., Yu, C","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.206953Z"},"links":{"cited_paper":"/paper/2403.13372","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:a0f95a1de78ec1d43f461795cfadb21bb69ed0791537d20ecde2b521479fa43a","observation_id":"e3738c32-c27f-4543-aa82-ea31e4711187","resolution":{"observed_at":"2026-08-03T00:13:30.206953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:30.321333Z","title":"E., and Zhou, W","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.321333Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:89d96044a3ed8ec3259c32ad54d2681e55069e297af6ba12fc1176e26c665ae8","observation_id":"9a4cb23e-93d5-4e7a-b7a3-f78585c61635","resolution":{"observed_at":"2026-08-03T00:13:30.321333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:30.397515Z","title":"Yes”, “No","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:30.397515Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:894f23c6e75144b074187fee8228e4d6cc8e4f1d40b1f4bf9183645991b3a642","observation_id":"0f994dda-d6a8-42fd-964a-a8595da18906","resolution":{"observed_at":"2026-08-03T00:13:30.397515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-03T00:13:25.873768Z","title":"P., Perelman, A., Ramesh, A., Clark, A., Ostrow, A., Welihinda, A., Hayes, A., Radford, A., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.873768Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:4259c846058d903d59aade345a34aa3cf88e17da19a8e93a9e2c3dfe7c3b0364","observation_id":"5a0a8997-cc09-4a0b-a013-6621dda5a016","resolution":{"observed_at":"2026-08-03T00:13:25.873768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.25140","last_updated":"2026-03-16T20:49:28Z","snapshot_observed_at":"2026-08-02T12:08:17.149184Z","submitted_at":"2025-09-29T17:51:03Z","title":"ReasoningBank: Scaling Agent Self-Evolving with Reasoning Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.25140","snapshot_observed_at":"2026-08-03T00:13:27.661295Z","title":"T., Daruki, S., Tang, X., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:27.661295Z"},"links":{"cited_paper":"/paper/2509.25140","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:e1c3f9fd8bebc49b429382342aca0c1cc2b9ce4e29d1173d7c50bcb11715494f","observation_id":"11359bac-b2ad-4c72-8e1a-64e34cd6f14f","resolution":{"observed_at":"2026-08-03T00:13:27.661295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.17032","last_updated":"2025-11-02T13:42:19Z","snapshot_observed_at":"2026-07-06T18:51:06.750136Z","submitted_at":"2024-07-24T06:35:05Z","title":"Gymnasium: A Standard Interface for Reinforcement Learning Environments","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.17032","snapshot_observed_at":"2026-08-03T00:13:28.882418Z","title":"U., De Cola, G., Deleu, T., Goul˜ao, M., Kallinteris, A., Krimmel, M., KG, A., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:28.882418Z"},"links":{"cited_paper":"/paper/2407.17032","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:aa69b7bcce4b2d96ac5563097b4f5437684d0478434d1eca7bce7a90624b87ef","observation_id":"8ce43951-02aa-4c32-9da1-fd317ea5175d","resolution":{"observed_at":"2026-08-03T00:13:28.882418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-03T00:13:26.072593Z","title":"Openai o1 system card.arXiv preprint arXiv:2412.16720,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:26.072593Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:73cbc2e7da806994dc1cb8de630031c0e0701b521ce5dee3822f49aa5468d70b","observation_id":"d3b8c592-c102-4c97-a2f1-41c7335db57c","resolution":{"observed_at":"2026-08-03T00:13:26.072593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T00:13:25.140784Z","title":"Behavior injection: Preparing language models for reinforcement learning.arXiv preprint arXiv:2505.18917,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:25.140784Z"},"links":{"citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:767eb4dd4174407ad21cd9ed2c6e7011f312ae578c8c58ee82220c377ddaf65d","observation_id":"f09582dc-a9eb-4fe9-9b35-23ec975f43e3","resolution":{"observed_at":"2026-08-03T00:13:25.140784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18449","last_updated":"2025-12-01T00:16:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-25T18:45:04Z","title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18449","snapshot_observed_at":"2026-08-03T00:13:29.208730Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization","version":2},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-03T00:13:29.208730Z"},"links":{"cited_paper":"/paper/2502.18449","citing_paper":"/paper/2602.11351"},"observation_digest":"sha256:6c375a53a09749bffa1e4b41f23bafd2941e8c4087d8ab7de6489486678e0287","observation_id":"a279625d-8e07-44d8-a331-6aec4580110f","resolution":{"observed_at":"2026-08-03T00:13:29.208730Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2602.11351","last_updated":"2026-06-28T23:49:44Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-03T00:13:22.529238Z","submitted_at":"2026-02-11T20:40:43Z","title":"Pushing Forward Pareto Frontiers of Proactive Agents with Behavioral Agentic Optimization"},"reference_resolution":{"displayed":41,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":41,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":41},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 41 of 41 outbound references and 0 inbound Pith citation observations for arXiv:2602.11351."}