{"as_of":"2026-08-06T02:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ad737905ca55976f2197f5afc103422f2953f9d8b2d205b98c7496282c590545","coverage":[{"denominator":33,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T23:52:53.822416Z","state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.15218/citation-record","integrity":"/paper/2607.15218/integrity","json":"/paper/2607.15218/citation-record.json","paper":"/paper/2607.15218"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.14219","last_updated":"2024-08-30T21:17:17Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T14:32:33Z","title":"Phi-3 Technical Report: A Highly Capable Language Model Locally on Your Phone","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14219","snapshot_observed_at":"2026-08-01T23:52:50.942964Z","title":"Phi-3 technical re- port: A highly capable language model locally on your phone.arXiv preprint arXiv:2404.14219,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:50.942964Z"},"links":{"cited_paper":"/paper/2404.14219","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:ccc95093238d83e887667859ec5406fbe9f247ef079611ce99caa482378c1242","observation_id":"1fc52820-4435-49d1-be5c-a05ae20aca2e","resolution":{"observed_at":"2026-08-01T23:52:50.942964Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.08073","last_updated":"2022-12-15T06:19:23Z","snapshot_observed_at":"2026-08-02T04:53:58.766070Z","submitted_at":"2022-12-15T06:19:23Z","title":"Constitutional AI: Harmlessness from AI Feedback","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.08073","snapshot_observed_at":"2026-08-01T23:52:51.152417Z","title":"Constitutional AI: Harmlessness from AI feedback.arXiv preprint arXiv:2212.08073,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.152417Z"},"links":{"cited_paper":"/paper/2212.08073","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:add8b24628684fdc7ec843ead2ef9134bf9f6f6ea3d0ee76c5aa58666e220bdb","observation_id":"99c3a7f3-b077-4896-a13a-3bcc5206d674","resolution":{"observed_at":"2026-08-01T23:52:51.152417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02737","last_updated":"2025-02-04T21:43:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-04T21:43:16Z","title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02737","snapshot_observed_at":"2026-08-01T23:52:51.225771Z","title":"SmolLM2: When smol goes big – data-centric training of a small language model.arXiv preprint arXiv:2502.02737,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.225771Z"},"links":{"cited_paper":"/paper/2502.02737","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:4774630d984067c8dec4e49b7e66638036d1d5d488fa27f03d77422902fb1c06","observation_id":"8e25b957-f304-4cbb-9607-cd4b12107e11","resolution":{"observed_at":"2026-08-01T23:52:51.225771Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.548149Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.548149Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:f62a9f50e2bcc52e08c2821a6775afc8fcd322aef95f06d81d658291601818b5","observation_id":"3c92058b-21ed-4451-9603-3beaa4b563dd","resolution":{"observed_at":"2026-08-01T23:52:51.548149Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05993","last_updated":"2024-09-11T14:42:29Z","snapshot_observed_at":"2026-07-06T17:57:31.912970Z","submitted_at":"2024-04-09T03:54:28Z","title":"AEGIS: Online Adaptive AI Content Safety Moderation with Ensemble of LLM Experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.05993","snapshot_observed_at":"2026-08-01T23:52:51.607980Z","title":"AEGIS: On- line adaptive AI content safety moderation with ensemble of LLM experts.arXiv preprint arXiv:2404.05993,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.607980Z"},"links":{"cited_paper":"/paper/2404.05993","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:ac1e8a2f077e5dbc12a1d20841e331a3098ce200b65a0737d3fd8eb880468e07","observation_id":"0a908996-fc3e-49fb-9ad0-651b6a13bb62","resolution":{"observed_at":"2026-08-01T23:52:51.607980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14650","last_updated":"2025-04-20T15:12:14Z","snapshot_observed_at":"2026-07-06T21:12:12.799570Z","submitted_at":"2025-04-20T15:12:14Z","title":"A Framework for Benchmarking and Aligning Task-Planning Safety in LLM-Based Embodied Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.14650","snapshot_observed_at":"2026-08-01T23:52:51.679091Z","title":"Language models as zero-shot planners: Extracting actionable knowledge for embodied agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.679091Z"},"links":{"cited_paper":"/paper/2504.14650","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:a4823130964488c8b6c68abf0c175ab5eb09f765ec3569fbfd96300fdc4d98de","observation_id":"3e103e25-3104-4283-b76d-e040d337b233","resolution":{"observed_at":"2026-08-01T23:52:51.679091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06674","last_updated":"2023-12-07T19:40:50Z","snapshot_observed_at":"2026-07-06T17:00:00.321552Z","submitted_at":"2023-12-07T19:40:50Z","title":"Llama Guard: LLM-based Input-Output Safeguard for Human-AI Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06674","snapshot_observed_at":"2026-08-01T23:52:51.813389Z","title":"Llama guard: LLM- based input-output safeguard for human-AI conversations.arXiv preprint arXiv:2312.06674,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.813389Z"},"links":{"cited_paper":"/paper/2312.06674","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:952377044e8527f142d46ee8482f33dd7e2e3ceb929a3d0a8ff5d4db29537176","observation_id":"b9897fd6-cab1-43d9-af9a-a2a9f498b85d","resolution":{"observed_at":"2026-08-01T23:52:51.813389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05474","last_updated":"2022-08-26T17:12:17Z","snapshot_observed_at":"2026-07-06T06:14:28.435222Z","submitted_at":"2017-12-14T23:17:24Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.05474","snapshot_observed_at":"2026-08-01T23:52:51.884733Z","title":"AI2-THOR: An interactive 3D environment for visual AI.arXiv preprint arXiv:1712.05474,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.884733Z"},"links":{"cited_paper":"/paper/1712.05474","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:9117033fa64b99a8cf7cdbfb2d4adcd3f5c0c308474ee18e7811750c58aba788","observation_id":"f307715a-0c52-4172-bcb3-d55374f2fb78","resolution":{"observed_at":"2026-08-01T23:52:51.884733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.980971Z","title":"AGENTSAFE: Benchmarking the safety of embodied agents on hazardous instructions.arXiv preprint arXiv:2506.14697,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.980971Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:c8672d3893d3acb73f95ff6fe11c9207b4ca968304e475ade37e899ea6113490","observation_id":"45d32663-42a5-4dc4-af10-23bdb3dd896e","resolution":{"observed_at":"2026-08-01T23:52:51.980971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:52.049845Z","title":"G-Eval: NLG evaluation using GPT-4 with better human alignment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.049845Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:ea12720fdb7254083bd43f867fdbd768dfbdb47a991876e97063a00de19d3941","observation_id":"04633014-3ac9-4779-ad91-b89bf0da1569","resolution":{"observed_at":"2026-08-01T23:52:52.049845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-01T23:52:52.190795Z","title":"The llama 3 herd of models.arXiv preprint arXiv:2407.21783,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.190795Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:5e20a44e363e1d5f7d7084cdd84027c8cc9a61f04b548fc438f397c20cb45048","observation_id":"99da4540-4375-4875-b54d-c7697ac349a4","resolution":{"observed_at":"2026-08-01T23:52:52.190795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:52.274400Z","title":"IS-Bench: Evaluating interactive safety of VLM-driven embodied agents in daily household tasks.arXiv preprint arXiv:2506.16402,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.274400Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:f1e7782cf6c11256b7c226fd5f65062601311333c0658cef7a0890eef4215419","observation_id":"e51a3e38-95ef-4544-b86e-85866116c1f4","resolution":{"observed_at":"2026-08-01T23:52:52.274400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06824","last_updated":"2024-08-19T01:18:41Z","snapshot_observed_at":"2026-07-06T16:30:37.867641Z","submitted_at":"2023-10-10T17:54:39Z","title":"The Geometry of Truth: Emergent Linear Structure in Large Language Model Representations of True/False Datasets","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06824","snapshot_observed_at":"2026-08-01T23:52:52.384452Z","title":"The geometry of truth: Emergent linear structure in LLM repre- sentations of true/false datasets.arXiv preprint arXiv:2310.06824,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.384452Z"},"links":{"cited_paper":"/paper/2310.06824","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:a0406f1dc0278f9c73e5ce41bd8e1ba2143eba8e6151599104aaf243ec36839e","observation_id":"3da758d5-4aee-449c-819c-4ebfcceacde1","resolution":{"observed_at":"2026-08-01T23:52:52.384452Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:52.538702Z","title":"Don’t let your robot be harmful: Responsible robotic manipulation via safety- as-policy.arXiv preprint arXiv:2411.18289,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.538702Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:ca8a5d8fc362ee010039603fbbeffb42001a6741a1cec795de429ad63a103b51","observation_id":"5683ea90-6138-47ea-bb6e-be781e9317ad","resolution":{"observed_at":"2026-08-01T23:52:52.538702Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06892","last_updated":"2025-03-10T03:37:36Z","snapshot_observed_at":"2026-07-06T20:49:37.096479Z","submitted_at":"2025-03-10T03:37:36Z","title":"SafePlan: Leveraging Formal Logic and Chain-of-Thought Reasoning for Enhanced Safety in LLM-based Robotic Task Planning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.06892","snapshot_observed_at":"2026-08-01T23:52:52.704841Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.704841Z"},"links":{"cited_paper":"/paper/2503.06892","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:0a2da77d9051dae0b69fa28c3bdfaa1f4ad92d8d6189a78ceb68eec1a46c492d","observation_id":"22e84c45-dd55-4ab4-baf2-a22da5c9bca9","resolution":{"observed_at":"2026-08-01T23:52:52.704841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-01T23:52:52.872352Z","title":"GPT-4 technical report.arXiv preprint arXiv:2303.08774,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:52.872352Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:0309606f4ec5072b3ed6658421cff0a31aa0e106787fbfe93f564b20b5b57576","observation_id":"24fd2da0-bf5b-4d9c-87f0-2bea544254ca","resolution":{"observed_at":"2026-08-01T23:52:52.872352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-01T23:52:53.302816Z","title":"Llama 2: Open founda- tion and fine-tuned chat models.arXiv preprint arXiv:2307.09288,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.302816Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:b725286aaddd348c8f97ed23bb01e1c34dd7580ad77b274a7c1b982f75aba835","observation_id":"d98b5c67-da3e-4355-bc3b-be410e503ddb","resolution":{"observed_at":"2026-08-01T23:52:53.302816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.10248","last_updated":"2024-10-10T13:20:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-20T12:21:05Z","title":"Steering Language Models With Activation Engineering","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.10248","snapshot_observed_at":"2026-08-01T23:52:53.373741Z","title":"Vazquez, Ulisse Mini, and Monte MacDiarmid","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.373741Z"},"links":{"cited_paper":"/paper/2308.10248","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:be7e06fc9d727440a98d13348c3931315eddbed27b444fb312b3e77912dd0d81","observation_id":"5ec62ffe-87e2-4754-ba2d-9e91bed1400e","resolution":{"observed_at":"2026-08-01T23:52:53.373741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16291","last_updated":"2023-10-19T16:27:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-25T17:46:38Z","title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16291","snapshot_observed_at":"2026-08-01T23:52:53.449275Z","title":"V oyager: An open-ended embodied agent with large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.449275Z"},"links":{"cited_paper":"/paper/2305.16291","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:d86d39bdc6efe074471be62d8c07885395c56f9c337b8642b17dcfb021504b78","observation_id":"1e1e2cb9-250e-4926-b811-27d758c4e5ab","resolution":{"observed_at":"2026-08-01T23:52:53.449275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15699","last_updated":"2025-06-19T04:21:00Z","snapshot_observed_at":"2026-07-06T21:12:55.678968Z","submitted_at":"2025-04-22T08:34:35Z","title":"Advancing Embodied Agent Security: From Safety Benchmarks to Input Moderation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.15699","snapshot_observed_at":"2026-08-01T23:52:53.514874Z","title":"Advancing embodied agent security: From safety benchmarks to input moderation.arXiv preprint arXiv:2504.15699,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.514874Z"},"links":{"cited_paper":"/paper/2504.15699","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:e8d1e92a4e468271ae1d56b21c248ce67d96665cff37c19c4e7982459dbe7073","observation_id":"b8684d64-8159-4a6d-a6ac-07c8750107e7","resolution":{"observed_at":"2026-08-01T23:52:53.514874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:53.574727Z","title":"SafeAgentBench: A benchmark for safe task planning of embodied LLM agents.arXiv preprint arXiv:2412.13178,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.574727Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:d26a540df2c34019601f75fec9016e0b9816eec1cb7bc590c9ee72e6ab197fd3","observation_id":"490fb07f-1faa-4fbf-abf3-7c9a68b1cd51","resolution":{"observed_at":"2026-08-01T23:52:53.574727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:53.629239Z","title":"R-Judge: Bench- marking safety risk awareness for LLM agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.629239Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:1821118962f816aa93611ea73cc69ae622ee148cdbb49a9dd8d5deb1ea042aa5","observation_id":"b77de546-c204-434f-b5c5-7b7d79ddacb7","resolution":{"observed_at":"2026-08-01T23:52:53.629239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21772","last_updated":"2024-08-04T22:13:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-31T17:48:14Z","title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21772","snapshot_observed_at":"2026-08-01T23:52:53.689569Z","title":"ShieldGemma: Generative AI content moderation based on gemma.arXiv preprint arXiv:2407.21772,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.689569Z"},"links":{"cited_paper":"/paper/2407.21772","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:d1d7033c9e2fb844ff6bd160211e548031573dcb2adf7b691e98b699adda93f8","observation_id":"bf547adc-49f9-4247-a269-2ae631839471","resolution":{"observed_at":"2026-08-01T23:52:53.689569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04449","last_updated":"2024-11-28T12:28:02Z","snapshot_observed_at":"2026-07-06T18:58:22.457329Z","submitted_at":"2024-08-08T13:19:37Z","title":"EARBench: Towards Evaluating Physical Risk Awareness for Task Planning of Foundation Model-based Embodied AI Agents","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04449","snapshot_observed_at":"2026-08-01T23:52:53.743894Z","title":"EARBench: Towards evaluating physical risk awareness for task planning of foundation model-based embod- ied AI agents.arXiv preprint arXiv:2408.04449,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.743894Z"},"links":{"cited_paper":"/paper/2408.04449","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:56d025897511d66b3dbdaf8073a9fc743ca3174f65c5b075d56b49e23040059d","observation_id":"02dd3a31-fca3-4fbf-8a7b-fb2e2aa900c6","resolution":{"observed_at":"2026-08-01T23:52:53.743894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01405","last_updated":"2025-03-03T06:14:14Z","snapshot_observed_at":"2026-07-06T16:26:38.284922Z","submitted_at":"2023-10-02T17:59:07Z","title":"Representation Engineering: A Top-Down Approach to AI Transparency","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01405","snapshot_observed_at":"2026-08-01T23:52:53.822416Z","title":"SDD”/“PGD","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.822416Z"},"links":{"cited_paper":"/paper/2310.01405","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:80ada91c1ffdeff5dc2492bef4813276df8c2abbf5859aaf65bdef5fe71429ae","observation_id":"1b5897e3-f3dd-407e-934d-6791ca3f3427","resolution":{"observed_at":"2026-08-01T23:52:53.822416Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.934761Z","title":"SafeText: A benchmark for exploring physical safety in language models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.934761Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:c3c0d34a5dc987b48816f48d1be59c19459fa29c1e885809f66cb036d9bcf16d","observation_id":"d07c7891-e54a-4e97-803f-9b342e3d8767","resolution":{"observed_at":"2026-08-01T23:52:51.934761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-01T23:52:53.025835Z","title":"Qwen2.5 technical report.arXiv preprint arXiv:2412.15115,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2018,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.025835Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:820b11964be97216c644d19099219845b7010f98a4ea90d1978d2a61df3a19c4","observation_id":"45d71624-eabe-449b-aca1-166d013169a5","resolution":{"observed_at":"2026-08-01T23:52:53.025835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.08663","last_updated":"2025-03-11T17:50:47Z","snapshot_observed_at":"2026-07-06T20:50:51.469809Z","submitted_at":"2025-03-11T17:50:47Z","title":"Generating Robot Constitutions & Benchmarks for Semantic Safety","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.08663","snapshot_observed_at":"2026-08-01T23:52:53.215213Z","title":"Gen- erating robot constitutions & benchmarks for semantic safety.arXiv preprint arXiv:2503.08663,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:53.215213Z"},"links":{"cited_paper":"/paper/2503.08663","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:c8bc8b98394755d5be00d00061405cbe1e8f04c0d03120ea43364456e397039d","observation_id":"7929156b-5393-4500-9fa6-acfdce7b5cea","resolution":{"observed_at":"2026-08-01T23:52:53.215213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-01T23:52:51.313791Z","title":"RT-2: Vision-language-action models transfer web knowledge to robotic control.arXiv preprint arXiv:2307.15818,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.313791Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:2804aee3fe343f1eb38bd7908539afb34bdc7f1fdda7d3aba2a66e377d87953c","observation_id":"d6268bef-04c6-4401-a31a-56d243b830a0","resolution":{"observed_at":"2026-08-01T23:52:51.313791Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1610.01644","last_updated":"2018-11-22T23:40:00Z","snapshot_observed_at":"2026-07-06T05:13:30.860932Z","submitted_at":"2016-10-05T20:59:01Z","title":"Understanding intermediate layers using linear classifier probes","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1610.01644","snapshot_observed_at":"2026-08-01T23:52:51.078723Z","title":"Understanding intermediate layers using linear classifier probes.arXiv preprint arXiv:1610.01644,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.078723Z"},"links":{"cited_paper":"/paper/1610.01644","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:b4a7fad0795d264e553c97694335bf41ffd34401b1aeb042571a43f217bd82c9","observation_id":"1af75e5c-fa88-41ee-8bc1-d41f74c9ba51","resolution":{"observed_at":"2026-08-01T23:52:51.078723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T23:52:51.422778Z","title":"SafeMind: Benchmarking and mitigating safety risks in embodied LLM agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.422778Z"},"links":{"citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:bc1336519cbdaee4a5ed7922ec43e1c12584cbe1c1d360dc5b2fcdbdf2ccbcd5","observation_id":"6ee6a447-c712-4375-98d0-1a3d8f75ca07","resolution":{"observed_at":"2026-08-01T23:52:51.422778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.01691","last_updated":"2022-08-16T16:06:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-04T17:57:11Z","title":"Do As I Can, Not As I Say: Grounding Language in Robotic Affordances","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.01691","snapshot_observed_at":"2026-08-01T23:52:51.006565Z","title":"Do as I can, not as I say: Grounding language in robotic affordances.arXiv preprint arXiv:2204.01691,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.006565Z"},"links":{"cited_paper":"/paper/2204.01691","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:417041bd2b3a1994e91a300ed96a1756d9349f61c232542b49750d692d7e6efc","observation_id":"01797a0e-f032-45d7-a546-b4681acb2661","resolution":{"observed_at":"2026-08-01T23:52:51.006565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16174","last_updated":"2026-06-01T15:28:47Z","snapshot_observed_at":"2026-07-06T20:40:57.414593Z","submitted_at":"2025-02-22T10:31:50Z","title":"Efficient LLM Moderation with Multi-Layer Latent Prototypes","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.16174","snapshot_observed_at":"2026-08-01T23:52:51.468348Z","title":"Efficient LLM moderation with multi-layer latent prototypes.arXiv preprint arXiv:2502.16174,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-01T23:52:51.468348Z"},"links":{"cited_paper":"/paper/2502.16174","citing_paper":"/paper/2607.15218"},"observation_digest":"sha256:2803c104a33447a3dea331100e3ca104cbe30c95aad62472bf64750c67cbe6c0","observation_id":"520e3158-13b0-453c-9911-a67b79e05fe0","resolution":{"observed_at":"2026-08-01T23:52:51.468348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.15218","last_updated":"2026-07-16T17:20:38Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-05T20:45:37.604475Z","submitted_at":"2026-07-16T17:20:38Z","title":"When Words Are Safe But Actions Kill: Probing Physical Danger Beyond Text Safety in Hidden-State Risk Space"},"reference_resolution":{"displayed":33,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":33,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":33},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 33 of 33 outbound references and 0 inbound Pith citation observations for arXiv:2607.15218."}