{"as_of":"2026-08-15T14:46:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:001e200b6219e8945e6b5e2cf18bdff577a27d9dba3a8fb3a788a8d7a57711a8","coverage":[{"denominator":79,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":79,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T04:24:51.705273Z","state":"measured"},{"denominator":79,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":79,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.10030/citation-record","integrity":"/paper/2608.10030/integrity","json":"/paper/2608.10030/citation-record.json","paper":"/paper/2608.10030"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.00374","last_updated":"2025-04-01T02:45:02Z","snapshot_observed_at":"2026-08-07T16:18:44.375411Z","submitted_at":"2025-04-01T02:45:02Z","title":"When Persuasion Overrides Truth in Multi-Agent LLM Debates: Introducing a Confidence-Weighted Persuasion Override Rate (CW-POR)","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.00374","snapshot_observed_at":"2026-08-14T04:24:50.620020Z","title":"When persuasion overrides truth in multi-agent LLM debates: Introducing a confidence-weighted persuasion override rate (CW-POR).arXiv preprint arXiv:2504.00374, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.620020Z"},"links":{"cited_paper":"/paper/2504.00374","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:6e0e8c6958848b65959deed5a7c5ea0933e6253cc914c0e4df033d18ec77a330","observation_id":"806f8c5d-9b70-4f4b-88fb-a8da2abe3c45","resolution":{"observed_at":"2026-08-14T04:24:50.620020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.630310Z","title":"Playing repeated games with large language models.Nature Human Behaviour, 9 (7):1380–1390, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.630310Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:229ebeaeecb508cae38c0a3fa81f6b8f432edf7e2cc15400214444582fdd7db2","observation_id":"3f91f124-6274-4282-9777-b4d640c82977","resolution":{"observed_at":"2026-08-14T04:24:50.630310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.19355","last_updated":"2026-05-22T12:13:16Z","snapshot_observed_at":"2026-08-13T19:54:25.865240Z","submitted_at":"2026-05-22T12:13:16Z","title":"Information Discernment in Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.19355","snapshot_observed_at":"2026-08-14T04:24:50.650726Z","title":"Information discernment in large language models.arXiv preprint arXiv:2607.19355, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.650726Z"},"links":{"cited_paper":"/paper/2607.19355","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a2bfb9e93dd5a4885af013197672d8c0992ce3c03e3467c7ec8a61d147d0c0d1","observation_id":"2bdf8d5d-94ec-4eeb-94f9-f3ded4844c98","resolution":{"observed_at":"2026-08-14T04:24:50.650726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.662840Z","title":"Jagadish, Or Duek, Ilan Harpaz-Rotem, Marie-Christine Khorsandian, Achim Burrer, Erich Seifritz, Philipp Homan, Eric Schulz, and Tobias R","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.662840Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9af0cace0d90b06eff0b9aaa8de28ae121c16e3697dff4606ae1ac15bbeeec50","observation_id":"1383f290-c18b-4775-a234-a5de68d7762c","resolution":{"observed_at":"2026-08-14T04:24:50.662840Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s44387-026-00122-1","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:52.069933Z","title":"Inducing state anxiety in llm agents reproduces human-like biases in consumer decision-making.npj Artificial Intelli- gence, 2(1):55, 2026","venue":null,"work_id":"63284da6-264e-4f94-bd06-edd6c4af25f6","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.681803Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:76535e28cf556bc9f5ba0a1c433a8b0244c2e49b539726abcf4be7b8ba8d59c3","observation_id":"b9f7c56a-1de2-41dd-a790-44c62cd96c29","resolution":{"observed_at":"2026-08-14T04:24:52.079152Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.707456Z","title":"Modelling monotonic effects of ordinal predictors in bayesian regression models.British Journal of Mathematical and Statistical Psychology, 73(3):420–451, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.707456Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:11a7ba42dc5662e9afa20b5b9915dbb00f941402f794eae0f69f58e3eb0683a8","observation_id":"589efa11-535a-46c5-ab18-e7595566622e","resolution":{"observed_at":"2026-08-14T04:24:50.707456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.208297Z","title":"I want to break free! persuasion and anti-social behavior of LLMs in multi-agent settings with social hierarchy.Transactions on Machine Learning Research, 2025","venue":null,"work_id":"7da81323-b647-473d-9aee-e25c66d44079","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.734948Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:15142d6f816cb90a0436b6ced7a357a41dae03becf32a4ece37a93e58d7c0c48","observation_id":"398791e0-1613-4354-a699-f0a353134c28","resolution":{"observed_at":"2026-08-14T04:24:57.217227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13995","last_updated":"2025-09-29T21:29:38Z","snapshot_observed_at":"2026-08-15T02:05:07.177530Z","submitted_at":"2025-05-20T06:45:17Z","title":"ELEPHANT: Measuring and understanding social sycophancy in LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13995","snapshot_observed_at":"2026-08-14T04:24:50.768438Z","title":"ELEPHANT: Measuring and understanding social sycophancy in LLMs.arXiv preprint arXiv:2505.13995, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.768438Z"},"links":{"cited_paper":"/paper/2505.13995","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a7284ff81083fb579c18107b7df5c2dbd5639ff5f7f7fbc4e67ff901fc0041bf","observation_id":"7f969a5d-d7cb-425c-9ec6-a42a81c81952","resolution":{"observed_at":"2026-08-14T04:24:50.768438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.174146Z","title":"A framework for studying AI agent behavior: Evidence from consumer choice experiments","venue":null,"work_id":"0a836c41-5cce-4db4-9714-a45b816cfa90","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.775423Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a1bdd7ca554d8d226211667699820d4c587a0313cd72d48429cbe4143d951d28","observation_id":"6d2f408b-7813-4c93-ac74-1f1b2ae51bc2","resolution":{"observed_at":"2026-08-14T04:24:57.194436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.21588","last_updated":"2025-05-27T12:12:56Z","snapshot_observed_at":"2026-08-08T08:32:12.026914Z","submitted_at":"2025-05-27T12:12:56Z","title":"Herd Behavior: Investigating Peer Influence in LLM-based Multi-Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.21588","snapshot_observed_at":"2026-08-14T04:24:50.787214Z","title":"Herd behavior: Investigating peer influence in LLM-based multi-agent systems.arXiv preprint arXiv:2505.21588, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.787214Z"},"links":{"cited_paper":"/paper/2505.21588","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:8241255ccfe1239c68974a25dc60bd912905eafc8f59af2f6398135c08e2cb8f","observation_id":"3090842a-7c45-4a9b-8bcd-2eaff2d0ef48","resolution":{"observed_at":"2026-08-14T04:24:50.787214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.808228Z","title":"Estimating the reproducibility of psychological science.Science, 349(6251):aac4716, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.808228Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:77156fb5eaeb77c866731d06b1a8eb59e0566d9a5b7c39f3bed3a0025802bff2","observation_id":"2e2a8155-da1a-41b1-aaa3-78681e11ed4b","resolution":{"observed_at":"2026-08-14T04:24:50.808228Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06613","last_updated":"2024-07-22T14:32:33Z","snapshot_observed_at":"2026-08-12T23:48:17.108802Z","submitted_at":"2024-06-07T00:28:43Z","title":"GameBench: Evaluating Strategic Reasoning Abilities of LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06613","snapshot_observed_at":"2026-08-14T04:24:50.818635Z","title":"GameBench: Evaluating strategic reasoning abilities of LLM agents.arXiv preprint arXiv:2406.06613, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.818635Z"},"links":{"cited_paper":"/paper/2406.06613","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:862c031d72d025a0cf72fd3375548f341486d414d1ebc4753d924d563b1424a0","observation_id":"4d37b749-d358-4223-a03d-1799a59a109e","resolution":{"observed_at":"2026-08-14T04:24:50.818635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.833112Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.833112Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:8ddba9fe6eb7b69c7cfdaca61734b21d70a20d95d3ad60d5d39868b2c5b9fd63","observation_id":"ba42dbdc-7d2f-433e-a75c-477158e5e5c1","resolution":{"observed_at":"2026-08-14T04:24:50.833112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.838455Z","title":"AI on my shoulder: Supporting emotional labor in front-office roles with an LLM-based empathetic coworker","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.838455Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:77eb38120e1c91a2a72d115b0306f75930c5053151e68096c85c2ef666a12761","observation_id":"520220c4-ba3c-430b-86f5-4a0174d83abf","resolution":{"observed_at":"2026-08-14T04:24:50.838455Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03053","last_updated":"2025-07-10T14:54:28Z","snapshot_observed_at":"2026-08-15T12:49:22.284676Z","submitted_at":"2025-06-03T16:33:47Z","title":"MAEBE: Multi-Agent Emergent Behavior Framework","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.03053","snapshot_observed_at":"2026-08-14T04:24:50.855799Z","title":"MAEBE: Multi-agent emergent behavior framework.arXiv preprint arXiv:2506.03053, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.855799Z"},"links":{"cited_paper":"/paper/2506.03053","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:b4d8dc0e8af5d8a59c93cc804bb2eb31e32f2a4495f68fcf52465b9250de6d8d","observation_id":"9d8698d2-df69-4860-a308-978e30e9d8c4","resolution":{"observed_at":"2026-08-14T04:24:50.855799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.146596Z","title":null,"venue":null,"work_id":"9d7970db-079f-4b56-98e4-5a72c89eda90","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.888062Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:5aaf83ca0f42c27577c637e42245cee595b8a8e5d4e8a93e2176b1bf98f5cb19","observation_id":"9080aba5-86f2-4623-889e-8048cb46cf05","resolution":{"observed_at":"2026-08-14T04:24:57.152833Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14093","last_updated":"2024-12-20T02:22:19Z","snapshot_observed_at":"2026-08-13T11:12:58.507759Z","submitted_at":"2024-12-18T17:41:24Z","title":"Alignment faking in large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14093","snapshot_observed_at":"2026-08-14T04:24:50.897678Z","title":"Bowman, and Evan Hubinger","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.897678Z"},"links":{"cited_paper":"/paper/2412.14093","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:dd8355d74c005e5552b41dee3acbc89cf3c7e5ee62e9a4fe10169eb5cd82fe20","observation_id":"de6249ec-9ec2-42e1-bc4b-11492f39cec6","resolution":{"observed_at":"2026-08-14T04:24:50.897678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:57.053246Z","title":"Bowman, and Sara Price","venue":null,"work_id":"51cd4740-3715-45a7-850d-2a5640b8715a","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.909446Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f76b8c8f97606c39da603b4d99a32f244a41a642b6898c84c5f487b91e0707b0","observation_id":"d3c4c112-9685-4687-a7e7-8b228e29ef31","resolution":{"observed_at":"2026-08-14T04:24:57.074507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/085713-1979","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.991086Z","title":"Deceptionbench: A comprehensive benchmark for AI deception behaviors in real-world scenarios","venue":null,"work_id":"0cc09f2e-afb3-4404-ac1c-f9ecd548513f","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.918468Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:6c93120acb742fe93513342a98bdcc9b7bd8d0e8fdc18803473aed83966077b8","observation_id":"ec3f4417-30c3-42eb-aa3c-4100b5de8991","resolution":{"observed_at":"2026-08-14T04:24:52.001349Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.935705Z","title":"Per- sonaLLM: Investigating the ability of large language models to express personality traits","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.935705Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:029abf5265c205ace1af5459380468fb9a6069024d3ce6517428f7110a7f27a5","observation_id":"d6087938-5963-4976-b564-e5554fb0b791","resolution":{"observed_at":"2026-08-14T04:24:50.935705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.961687Z","title":"FollowBench: A multi-level fine-grained constraints following benchmark for large language models","venue":null,"work_id":"fc00fc3f-5c77-4150-9951-ef11339578f6","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.960570Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:4b2cde9d6dff141b9d28d253ce84026e70c2dfc37c7d90b485e95c994988e5aa","observation_id":"7d6b2803-1854-4015-9d15-164d444ec423","resolution":{"observed_at":"2026-08-14T04:24:57.004879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.970341Z","title":"Can large language models be good emotional supporter? miti- gating preference bias on emotional support conversation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.970341Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a28d5967a40e920e41ce2ccbbb6d772edef1f460b2ab76c4dc3a3f668d85bde0","observation_id":"22d42a15-f3ac-4ab0-b402-1240c5d36cfb","resolution":{"observed_at":"2026-08-14T04:24:50.970341Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0855.38186","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:53.659985Z","title":"Toward a science of ai agent societies","venue":null,"work_id":"90aeadbf-01bb-43b5-9ecf-380b41306256","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.981263Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:11a68d12f7610251175de1a9be5b39ac555ec60e626cc6b61e4bdafd8d6b9beb","observation_id":"cb015e79-0394-480b-820b-ca9b3ed51240","resolution":{"observed_at":"2026-08-14T04:24:53.734753Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.880188Z","title":"Aerobat code and data repository","venue":null,"work_id":"30ee91e7-4358-400e-96ac-983c8814a000","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.002229Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:31c9a95e38edc5d5d09334b93e283783c3916deae32ad06298ad3dc175920b26","observation_id":"2aea2229-0122-44f5-9154-07e1783ab6ae","resolution":{"observed_at":"2026-08-14T04:24:56.924832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2504.08016","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:53.461417Z","title":"Emergence of psychopathological computations in large language models.arXiv preprint arXiv:2504.08016, 2025","venue":null,"work_id":"0cdfd57d-8627-49a5-8b4c-2262547d5d82","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.017300Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0fee3972db889fade90ead0934e80a50510a7391a65ca51529f50491d29412b0","observation_id":"4f1c2a1f-4b68-45c4-9e1b-907c4ffd9409","resolution":{"observed_at":"2026-08-14T04:24:53.514750Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.18148","last_updated":"2024-03-26T23:14:34Z","snapshot_observed_at":"2026-08-13T00:44:17.396229Z","submitted_at":"2024-03-26T23:14:34Z","title":"Large Language Models Produce Responses Perceived to be Empathic","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.18148","snapshot_observed_at":"2026-08-14T04:24:51.023322Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.023322Z"},"links":{"cited_paper":"/paper/2403.18148","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7a806f90b481fd33df4bd309178676429a371d0021f5ec4fa586960c57db5a81","observation_id":"0758d6c4-d89c-4260-98c9-197426603430","resolution":{"observed_at":"2026-08-14T04:24:51.023322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.05622","last_updated":"2026-07-09T05:22:20Z","snapshot_observed_at":"2026-08-06T18:01:48.020763Z","submitted_at":"2026-06-04T02:47:29Z","title":"AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.05622","snapshot_observed_at":"2026-08-14T04:24:51.036759Z","title":"Fung, and Heng Ji","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.036759Z"},"links":{"cited_paper":"/paper/2606.05622","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:02716d285e4f95298df1d01ab5b50a9608dda4915671ff32782b4aca8aec14a0","observation_id":"3910385c-8171-4aa2-8c0f-38df0153502a","resolution":{"observed_at":"2026-08-14T04:24:51.036759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.820395Z","title":"Strategic behavior of large language models and the role of game structure versus contextual framing.Scientific Reports, 14(1):18490, 2024","venue":null,"work_id":"d78e0766-7eb7-43c9-aae1-1c30e61cec8a","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.043242Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:dbffb9a96fc01bd8fc398f8bbc258a124f4f2c3fb90f8809e6d94c6d518d818a","observation_id":"2761bdd5-9235-4f6f-b9aa-bcc4aa24a158","resolution":{"observed_at":"2026-08-14T04:24:56.836787Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.055176Z","title":"Agentic misalignment: How LLMs could be insider threats.arXiv preprint arXiv:2510.05179, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.055176Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:dcd5111303ffdf2b4b3d67d1ddce004a734d62a0c21183099399ba8081a10e86","observation_id":"de1900e2-e63a-4c5f-9140-222913df7cff","resolution":{"observed_at":"2026-08-14T04:24:51.055176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.11794","last_updated":"2024-04-25T01:34:42Z","snapshot_observed_at":"2026-08-13T00:27:48.434842Z","submitted_at":"2024-04-17T23:02:43Z","title":"Automated Social Science: Language Models as Scientist and Subjects","version":2},"cited_work":{"arxiv_id":"2404.11794","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.11794","snapshot_observed_at":"2026-08-14T04:24:53.113220Z","title":"Automated Social Science: Language Models as Scientist and Subjects","venue":"econ.GN","work_id":"697924e6-cc77-4fe4-b1ec-898b1d092a0a","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.083853Z"},"links":{"cited_paper":"/paper/2404.11794","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:997ea6c07734b03031b04b4ccf1a61cace634dd488b9487cdd5cfaf2054047f4","observation_id":"7a8ddb43-761a-469a-a98d-fac0259739d5","resolution":{"observed_at":"2026-08-14T04:24:53.121067Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.796683Z","title":"ALYMPICS: LLM agents meet game theory","venue":null,"work_id":"0424e9ab-7352-4a40-9a2e-1994df81b24c","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.094814Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:c175156fa51702f4e1cf384351da7352c89a464bc2b5158c9f05c10e2df65aaa","observation_id":"aa27b351-58c1-40e6-933a-7a38a1cbc1f4","resolution":{"observed_at":"2026-08-14T04:24:56.802910Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04984","last_updated":"2025-01-14T20:16:01Z","snapshot_observed_at":"2026-08-15T04:02:05.429942Z","submitted_at":"2024-12-06T12:09:50Z","title":"Frontier Models are Capable of In-context Scheming","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04984","snapshot_observed_at":"2026-08-14T04:24:51.169617Z","title":"Frontier models are capable of in-context scheming.arXiv preprint arXiv:2412.04984, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.169617Z"},"links":{"cited_paper":"/paper/2412.04984","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:fde49f7919ed42c80cad08235ab5d7004418e2cd5320dfe91ebca8e8fc276502","observation_id":"16488ed7-a5e9-45d3-9285-93e436a9ce87","resolution":{"observed_at":"2026-08-14T04:24:51.169617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.184477Z","title":"Learn- ing when to plan: Efficiently allocating test-time compute for LLM agents.arXiv preprint arXiv:2509.03581, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.184477Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0e96372a34aa1e6629eff62193711c7ebb3717b3d1699f2dbaf7c2139e514d3f","observation_id":"ca4a715c-b8a7-4d7d-b8a6-dab97ba3b988","resolution":{"observed_at":"2026-08-14T04:24:51.184477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.12920","last_updated":"2025-08-18T13:40:10Z","snapshot_observed_at":"2026-08-05T19:08:39.397422Z","submitted_at":"2025-08-18T13:40:10Z","title":"Do Large Language Model Agents Exhibit a Survival Instinct? An Empirical Study in a Sugarscape-Style Simulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.12920","snapshot_observed_at":"2026-08-14T04:24:51.134751Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.134751Z"},"links":{"cited_paper":"/paper/2508.12920","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:22ecda98507d1e3c9ed30b26b8e456909cee30c2a69075b925f39d213162c495","observation_id":"de270470-c86f-4b6b-907b-2c126c39c5a9","resolution":{"observed_at":"2026-08-14T04:24:51.134751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.210759Z","title":"O’Brien, Carrie Jun Cai, Meredith Ringel Morris, Percy Liang, and Michael S","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.210759Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:443654005d4733baf6e791b5795fb07c9aaacb30fdcc907d4b0e3969d8030a09","observation_id":"1357b927-ecfa-4fdd-83bc-7deb79d0dd03","resolution":{"observed_at":"2026-08-14T04:24:51.210759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10109","last_updated":"2026-06-28T22:37:51Z","snapshot_observed_at":"2026-08-13T12:33:17.230204Z","submitted_at":"2024-11-15T11:14:34Z","title":"LLM Agents Grounded in Self-Reports Enable General-Purpose Simulation of Individuals","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10109","snapshot_observed_at":"2026-08-14T04:24:51.223542Z","title":"Zou, Jonne Kamphorst, Niles Egan, Aaron Shaw, Benjamin Mako Hill, Carrie Cai, Meredith Ringel Morris, Percy Liang, Robb Willer, and Michael S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.223542Z"},"links":{"cited_paper":"/paper/2411.10109","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:34f0eee5c87282f462de97ea88594a8672d433b332c09b6cd1b8a4fdd892238d","observation_id":"f990cd49-3043-48d9-94a1-2479d0ba398c","resolution":{"observed_at":"2026-08-14T04:24:51.223542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.728566Z","title":"Do the rewards justify the means? Mea- suring trade-offs between rewards and ethical behavior in the MACHIA VELLI benchmark","venue":null,"work_id":"0c521912-5879-4db5-a442-a24524fbfe0a","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.197682Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:e93bebbf5509cdf949dc502d13ef97011238d120972309ef7997fe97522fca62","observation_id":"5688f647-3349-4f16-97d9-bcc826e0b5c7","resolution":{"observed_at":"2026-08-14T04:24:56.752116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2307/jj.6380610.6","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.925075Z","title":"Psychological predicates","venue":null,"work_id":"dfc90046-d094-4bfa-bace-db4c8c70704d","year":1967},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.246184Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:63f1ddd49f9f31a369dbf6b9bf1b8501d38c02814731f84c76e1d3c733e71345","observation_id":"8ed6ebf5-41f9-4048-a640-6c2fc72ffb50","resolution":{"observed_at":"2026-08-14T04:24:51.945629Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16944","last_updated":"2025-05-22T17:31:10Z","snapshot_observed_at":"2026-08-14T00:08:05.335165Z","submitted_at":"2025-05-22T17:31:10Z","title":"AGENTIF: Benchmarking Instruction Following of Large Language Models in Agentic Scenarios","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.16944","snapshot_observed_at":"2026-08-14T04:24:51.262784Z","title":"AGENTIF: Benchmarking instruction following of large language models in agentic scenarios","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.262784Z"},"links":{"cited_paper":"/paper/2505.16944","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:4b1e77639f83b405b5c78cb16c2249815a27e12644100c0615429e06ec4820ec","observation_id":"f629dd28-7fca-4bb6-bfeb-e50c5be9f953","resolution":{"observed_at":"2026-08-14T04:24:51.262784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.08691","last_updated":"2026-04-10T15:11:22Z","snapshot_observed_at":"2026-08-13T05:19:36.652601Z","submitted_at":"2025-02-12T15:27:07Z","title":"AgentSociety: Large-Scale Simulation of LLM-Driven Generative Agents Advances Understanding of Human Behaviors and Society","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.08691","snapshot_observed_at":"2026-08-14T04:24:51.231665Z","title":"Agentsociety: Large-scale simulation of llm-driven generative agents advances understanding of human behaviors and society.arXiv preprint arXiv:2502.08691, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.231665Z"},"links":{"cited_paper":"/paper/2502.08691","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0182159bbdd6ec18755adf37ab73e4dd89462f4dfbf0154b62f44c240ef3b653","observation_id":"1894fb4f-512f-44da-af36-38e3a6ff7b6b","resolution":{"observed_at":"2026-08-14T04:24:51.231665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.19299","last_updated":"2026-07-21T16:41:14Z","snapshot_observed_at":"2026-08-06T00:04:08.698733Z","submitted_at":"2025-10-22T07:00:33Z","title":"Learning to Make Friends: Coaching LLM Agents toward Emergent Social Ties","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.19299","snapshot_observed_at":"2026-08-14T04:24:51.289596Z","title":"Schneider, Lin Tian, and Marian-Andrei Rizoiu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.289596Z"},"links":{"cited_paper":"/paper/2510.19299","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:490e66a3854ba4a249043aeb7f0b6811e5f86c4afee7a20d0cade5091f90de79","observation_id":"eea0eac9-a4f7-4426-afe8-dd43e0374962","resolution":{"observed_at":"2026-08-14T04:24:51.289596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.643642Z","title":"Bowman, Newton Cheng, Esin Durmus, Zac Hatfield-Dodds, Scott R","venue":null,"work_id":"62af5dc0-cd51-4d9e-833e-b3f8e6108b8e","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.299621Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9b59b6e212ee8b02d88f507527d684baa390d530a36c356546f5161d6e2036ef","observation_id":"09261913-47c0-455a-b080-d88f07b02ffe","resolution":{"observed_at":"2026-08-14T04:24:56.673074Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:52.690918Z","title":"Escalation risks from language models in military and diplomatic decision-making","venue":null,"work_id":"12e854bb-5868-4253-9b9a-799b893c8846","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.275476Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:97b841f6a4fa7eb4e2ff88cf7b2a325448f931047cfc5458e6351196b4bb582c","observation_id":"eea89723-4058-4099-b1b3-ab29b49337ff","resolution":{"observed_at":"2026-08-14T04:24:52.706008Z","resolver_source":"arxiv_id_nonexistent","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.313872Z","title":"LLMs can’t handle peer pressure: Crumbling under multi-agent social interactions.arXiv preprint arXiv:2508.18321, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.313872Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:cf83ad2f4191e4bcc8695a1bd09df2b723cb572a0b366680690a212ef4dffe89","observation_id":"4f0ce941-dceb-4b20-a3aa-9568008f1205","resolution":{"observed_at":"2026-08-14T04:24:51.313872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/085713-0320","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.908151Z","title":"AI-Researcher: Autonomous sci- entific innovation","venue":null,"work_id":"64ce8f80-ee2b-4e02-9b08-19e12aead409","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.327559Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:56134658c4ca49cb9fd9a2cdc2d39bc366b37fa022cb74111537649f3937e903","observation_id":"f1187513-89ec-4a60-ba0a-cd987907d995","resolution":{"observed_at":"2026-08-14T04:24:51.913415Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.554236Z","title":"Reflexion: Language agents with verbal reinforcement learning","venue":null,"work_id":"21d028f3-1aa6-4820-b1ec-bd01a2b524b2","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.305846Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:dea4491a0d8449b44770712972e95872baac5077772088dacceecab52dd273e1","observation_id":"630ac671-dc36-4080-9dd3-665ffad888b9","resolution":{"observed_at":"2026-08-14T04:24:56.565613Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13208","last_updated":"2024-04-19T22:55:23Z","snapshot_observed_at":"2026-08-11T23:45:02.178667Z","submitted_at":"2024-04-19T22:55:23Z","title":"The Instruction Hierarchy: Training LLMs to Prioritize Privileged Instructions","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13208","snapshot_observed_at":"2026-08-14T04:24:51.367266Z","title":"The instruction hierarchy: Training LLMs to prioritize privileged instructions.arXiv preprint arXiv:2404.13208, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.367266Z"},"links":{"cited_paper":"/paper/2404.13208","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:d71d240f94f948b724d111e2dc6494eda5ce14eb8811f3365738d5d60a8a8ffb","observation_id":"707a48fa-9508-4b3a-a68e-0db47594eca5","resolution":{"observed_at":"2026-08-14T04:24:51.367266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.26100","last_updated":"2026-05-14T06:48:52Z","snapshot_observed_at":"2026-08-14T10:53:26.197154Z","submitted_at":"2025-09-30T11:20:41Z","title":"AgenticEval: Toward Agentic and Self-Evolving Safety Evaluation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2509.26100","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.26100","snapshot_observed_at":"2026-08-14T04:24:52.436114Z","title":"AgenticEval: Toward Agentic and Self-Evolving Safety Evaluation of Large Language Models","venue":"cs.AI","work_id":"902172b7-430c-4189-81ab-6a63d8a686e5","year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.374296Z"},"links":{"cited_paper":"/paper/2509.26100","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0cf53d2b91b5fc05b59b01336d66996e652581ce9820f6440e6ae0457eded262","observation_id":"7a088685-c2c3-4080-8e39-4c744a1ed8ce","resolution":{"observed_at":"2026-08-14T04:24:52.456823Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/075280-1693","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.858687Z","title":"PlanBench: An extensible benchmark for evaluating large language models on planning and reasoning about change","venue":null,"work_id":"3e58851e-1658-4b9a-8d6c-d080bc4eea16","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.358957Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f0e5d180b40b2e94dd6a66be129a288030104012ba24b4525e803fc850669cbb","observation_id":"8ac5d6e4-2adc-4270-8de5-5ea26314b28c","resolution":{"observed_at":"2026-08-14T04:24:51.865352Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.478355Z","title":"ReAct: Synergizing reasoning and acting in language models","venue":null,"work_id":"25006ca3-7fbf-4901-993f-e71002012d52","year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.414542Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1be8c3379f58a802155bbbcf6947a73e7fba80e1d16a9c618bd161872271a347","observation_id":"de57500b-2f34-4463-bfa5-c90649a8dfc2","resolution":{"observed_at":"2026-08-14T04:24:56.514806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.372493Z","title":"Position: Llms can’t jump","venue":null,"work_id":"3edac2ed-ea85-49c7-beef-fd26bbe7c58b","year":2026},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.428119Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:02c3aa4a32faa0846b164c7c17ac059e2986185b846870bef2f0032193946d9c","observation_id":"279df0e6-a3fb-4d35-9dc8-da62dba5db4e","resolution":{"observed_at":"2026-08-14T04:24:56.397207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.389322Z","title":"Nuclear deployed!: Analyzing catastrophic risks in decision-making of autonomous LLM agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.389322Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0aee01d13e964b3f22431dac23fc9395d42edfd24c900492d143ffd3341c8106","observation_id":"10c38f9d-123e-47e1-854d-4b1e0ea730f0","resolution":{"observed_at":"2026-08-14T04:24:51.389322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10157","last_updated":"2025-07-15T11:14:36Z","snapshot_observed_at":"2026-08-07T16:05:55.074747Z","submitted_at":"2025-04-14T12:12:52Z","title":"SocioVerse: A World Model for Social Simulation Powered by LLM Agents and A Pool of 10 Million Real-World Users","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10157","snapshot_observed_at":"2026-08-14T04:24:51.439987Z","title":"SocioVerse: A world model for social simulation powered by LLM agents and a pool of 10 million real-world users.arXiv preprint arXiv:2504.10157, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.439987Z"},"links":{"cited_paper":"/paper/2504.10157","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f1b2184a0bfabddd56aa7e474addbbfc061b560bdc32b57b12200273dbf41a1e","observation_id":"96d0a874-9369-49dd-9215-285e988602e1","resolution":{"observed_at":"2026-08-14T04:24:51.439987Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.256082Z","title":"CompeteAI: Understanding the competition dynamics in large language model-based agents","venue":null,"work_id":"e6a11d7f-e6b2-44cf-b334-730635f17a4d","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.447353Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:339e079ed5688639253091bdd5af4e9913c65fa6aace2bc8cd80d60394d408e7","observation_id":"58af1bbd-c493-466b-a407-cbf27754ac13","resolution":{"observed_at":"2026-08-14T04:24:56.283155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.433413Z","title":"Dive into the agent matrix: A realistic evaluation of self-replication risk in LLM agents.arXiv preprint arXiv:2509.25302, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.433413Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1310d854199fb3f1448b62f37395e7538e24ef8af8a641eb1ffd8566acba8701","observation_id":"f8935cfe-df32-4625-9ddf-21621dd3225c","resolution":{"observed_at":"2026-08-14T04:24:51.433413Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.079466Z","title":"Navigating the grey area: How expres- sions of uncertainty and overconfidence affect language models","venue":null,"work_id":"fb82c4f1-698d-439d-8076-d881e38f68df","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.461235Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:9bad429d39ff792205befdd1038c28e16b1d3d7046e57532bb4ecdb3a75568b6","observation_id":"d47b118f-08bc-483e-aae7-c6973334a775","resolution":{"observed_at":"2026-08-14T04:24:56.104750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.019924Z","title":"SOTOPIA: Interactive evaluation for social intelligence in language agents","venue":null,"work_id":"dbafac9c-74e4-41d4-bcfe-72dddabb8687","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.490605Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:191a7036fda2bdfca1fa1a2bae9a6e11f4e12ad01bfdd35a1acbba9a6c769d1e","observation_id":"c5faa021-d339-42ea-944e-9c2c204c5ae5","resolution":{"observed_at":"2026-08-14T04:24:56.038303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:56.161384Z","title":"ALI- Agent: Assessing LLMs’ alignment with human values via agent-based evaluation","venue":null,"work_id":"a06ea18a-b00d-42f2-9934-7fdb0e085370","year":2024},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.455534Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1ccd6b301a5b117fd9db0acbca4d0850066f0d46ab87e4c71979adba0672ca5d","observation_id":"7f6bfdfb-3093-4560-a4f0-5e6c5d31f381","resolution":{"observed_at":"2026-08-14T04:24:56.204748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.498858Z","title":"positive","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.498858Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:c0b7009c0a3b6ef95f4ccd4dabbb53bd176e5bedf58e5e413c619afbe346cdc4","observation_id":"2ad78825-ebb6-4224-a4d9-95249781abbf","resolution":{"observed_at":"2026-08-14T04:24:51.498858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.990517Z","title":null,"venue":null,"work_id":"8c35f8a6-b2e1-4073-b852-266315c5a2b4","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.505375Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:45b6acde842cbecde16adc21c0f219a70e62096cf89793217ca593c525a62613","observation_id":"0cb612de-3ee3-43a6-ac86-f65e1c69f969","resolution":{"observed_at":"2026-08-14T04:24:55.998608Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.946235Z","title":null,"venue":null,"work_id":"7693fb9d-10c8-42f3-9fa4-fa175eac1ef3","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.527886Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:1ccfd745b08cecb9c9f996ae16b09f42913227174737e3255c2ae0dbddf2d8d7","observation_id":"63178901-1947-41f3-afa7-219c02df16b6","resolution":{"observed_at":"2026-08-14T04:24:55.957704Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.827805Z","title":null,"venue":null,"work_id":"9ad8a812-5c43-4fd8-9e98-f3b14bd56e88","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.539063Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7183283e4d2f0e3c37fd460fcb71bbafef0ddeb0c70ebc74e5f1dcc49b006f1b","observation_id":"79b11712-6082-4da8-a248-79072d28f17f","resolution":{"observed_at":"2026-08-14T04:24:55.845708Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.748452Z","title":"None\" - task 2.exemption: if no meaningful interactions exist, write","venue":null,"work_id":"6037508f-775c-4186-90ba-e5309c2fe759","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.551241Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:779eeb8178443d7a77643fe7d8c66f6f112bf7d162ee492048e2367177494397","observation_id":"a98993bd-d96d-4aee-a24d-61b8281d49a2","resolution":{"observed_at":"2026-08-14T04:24:55.774750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.681443Z","title":"objective","venue":null,"work_id":"b390e81b-8060-4953-b40a-a39c1523ec28","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.562766Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7c515c0935ac0f8c41730e989d757e6e6ab66d7c7d8ac7ea1a3e577837e2ff04","observation_id":"796143db-41f4-4357-87d3-3541732f19a9","resolution":{"observed_at":"2026-08-14T04:24:55.724756Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.575676Z","title":null,"venue":null,"work_id":"a1e5139f-5e11-457d-aff2-784f09d39fbe","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.576142Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:f5f6500dd5584af0e7a5651934f9ddd0f8c51a609fda40482d31327a23a889fe","observation_id":"b31cc0aa-60b4-4907-882d-9e08e796def3","resolution":{"observed_at":"2026-08-14T04:24:55.587374Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.449469Z","title":null,"venue":null,"work_id":"5c8a60a6-6ac6-418d-a330-47a788db5f0b","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.587965Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:8d05e2a375b8879916f86496afc4e2adf06ef68b312ffe7ef5360e815098fd8b","observation_id":"e737a7e3-1fc8-4595-b7f7-2cffc0d210f4","resolution":{"observed_at":"2026-08-14T04:24:55.456910Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.384958Z","title":null,"venue":null,"work_id":"6b63f2e3-475d-4197-8b99-ed8e98fba724","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.593204Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:e0f38d1635f06a73777608962cbbaf77d69392803230ff96f63420ad8da5318f","observation_id":"83c7e6af-a54f-4f71-a285-5d6b661e6c5b","resolution":{"observed_at":"2026-08-14T04:24:55.420261Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.340100Z","title":null,"venue":null,"work_id":"cab208e5-d095-4750-bda3-9f5a2f87934c","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.602457Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:06285e877e6eb04333fe46a2d4b0494dc78023c87df8800530612dce3a287d74","observation_id":"57a3bb62-e5a9-4c94-a41a-86fea870700e","resolution":{"observed_at":"2026-08-14T04:24:55.356918Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.224752Z","title":null,"venue":null,"work_id":"762c9080-7dcd-467c-a71f-a8adf24ec75e","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.608105Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:634fe633736f05a5c9a7c420ea0ab143c652eac4c1c9cbf39f1ea348a84e7e91","observation_id":"63a2ead3-1e7a-484e-a3ad-a607ea74e511","resolution":{"observed_at":"2026-08-14T04:24:55.263910Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:55.119780Z","title":null,"venue":null,"work_id":"ad65dac8-f9cd-4de0-8740-e95e5f5ce671","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.613122Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:0ed146463cf31596152350aee8a5f191e5f07b897488392ef2679572ad9ba6f6","observation_id":"ab3f72f3-6805-4505-93b0-3100bda10edc","resolution":{"observed_at":"2026-08-14T04:24:55.164740Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.975365Z","title":null,"venue":null,"work_id":"d6e3f697-394a-4578-b10a-f66811493d9d","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.628160Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:45be510338c62980d81ca2ada1b0b716c1cb1081641217832458af4c154c3199","observation_id":"55a66085-e2bc-485c-97c4-c102dfa0f578","resolution":{"observed_at":"2026-08-14T04:24:55.003909Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.884511Z","title":null,"venue":null,"work_id":"5ca02665-585d-4e06-83a5-d98b18ee9c67","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.641847Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:37eb46ff0290b899082608c5778b2fbdb47b8f8e6f0141ce085ec6dce685b6c8","observation_id":"2489b948-bde9-433f-97e1-c3443642ead3","resolution":{"observed_at":"2026-08-14T04:24:54.914743Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.794765Z","title":null,"venue":null,"work_id":"79665977-0f00-4917-8d94-7a39b2a5e903","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.654911Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:97b68a09049d7b6a40600ad78f6e5f8637bba5b7f7bee230a91c20daca7becee","observation_id":"79d9f9ad-0c2a-4617-8e7f-773e8b4a5347","resolution":{"observed_at":"2026-08-14T04:24:54.832253Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.714837Z","title":null,"venue":null,"work_id":"9d8d1393-da75-4441-9b1a-a4e9b6666091","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.671098Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:7aeb7dc9df9207a141214a0d99feee147e07b75c747a2cf27df2403e7ef61e7c","observation_id":"036d3d6e-2688-46fc-85ea-2036f33388d3","resolution":{"observed_at":"2026-08-14T04:24:54.738686Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.640641Z","title":null,"venue":null,"work_id":"ad801c3e-3afc-416d-9f88-8e1901babbea","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.684880Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:5b1b0a3c214e41e6b0dfc1a28cbd6bbf5ceaf5b003ecc4dd263154d907366d50","observation_id":"a45e26c0-2160-40b6-8593-24de377c6026","resolution":{"observed_at":"2026-08-14T04:24:54.669376Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:54.569991Z","title":"In each subplot, its x-axis and y-axis ticks denote distinct evidence classes defined in its rubric yrubric","venue":null,"work_id":"ae34623d-ba24-47f1-b887-fc94f3761ea5","year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.705273Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:c27d448f30bcd01d091a71282f8ffa56dc2c67411ce6381e24236f0bd1945b80","observation_id":"54e24820-b0fa-41aa-939a-398b2940c9c4","resolution":{"observed_at":"2026-08-14T04:24:54.586598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:51.480581Z","title":"URL https://aclanthology.org/2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.480581Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:35d4ae5dffd40baeb407de6e806a399e3152373aeb9399eb196ca92c10dfab20","observation_id":"e0f9a378-7f09-4c59-83eb-1cb5eb113234","resolution":{"observed_at":"2026-08-14T04:24:51.480581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T04:24:50.879313Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:50.879313Z"},"links":{"citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:a5c3caaba4b8a883c57f82d8f7efdbbe67817c5883a8701aea6a40bc362f2e84","observation_id":"8e12f268-f866-4c19-8d1a-7c81374ad9fa","resolution":{"observed_at":"2026-08-14T04:24:50.879313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.13433","last_updated":"2026-05-29T00:45:48Z","snapshot_observed_at":"2026-08-03T09:37:04.326965Z","submitted_at":"2026-01-19T22:37:30Z","title":"Who Endorsed It? Measuring Authority Bias Across Expertise Levels in Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.13433","snapshot_observed_at":"2026-08-14T04:24:51.073476Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-14T04:24:51.073476Z"},"links":{"cited_paper":"/paper/2601.13433","citing_paper":"/paper/2608.10030"},"observation_digest":"sha256:b60ec8b82b5491b9ee6daedfd4eec5db252a75ba93d3781e9968b42eceda1ed2","observation_id":"de35ca0b-7e3e-46df-966c-24d164b8c7e3","resolution":{"observed_at":"2026-08-14T04:24:51.073476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.10030","last_updated":"2026-08-10T01:14:31Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-15T05:14:36.359460Z","submitted_at":"2026-08-10T01:14:31Z","title":"Automating and Scaling Behavioral Scientific Research on AI Agents"},"reference_resolution":{"displayed":79,"state_counts":{"malformed_identifier":3,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":50,"verified_exact":8,"verified_fuzzy":16},"total_outbound_references":79},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 79 of 79 outbound references and 0 inbound Pith citation observations for arXiv:2608.10030."}