{"as_of":"2026-08-20T05:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4b1f5014feea2aae71f3bf27ae96e7b9322071bed1a7bf26b70adc1f986741fd","coverage":[{"denominator":31,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T00:37:59.145162Z","state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.08029/citation-record","integrity":"/paper/2608.08029/integrity","json":"/paper/2608.08029/citation-record.json","paper":"/paper/2608.08029"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.668294Z","title":"Safety Beyond the Interface: Detecting Harm via Latent States in Large Language Models , year=","venue":null,"work_id":"ee5ea9ab-3240-4f29-befb-311ba874205f","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.012396Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:eb84d06df83d5114ea4fccfc6b151f838049c38ad265004ff8ff2f1784a8a8e1","observation_id":"6aa3591f-baa5-4be8-9d5b-eddc15455b21","resolution":{"observed_at":"2026-08-12T00:37:59.672979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.654798Z","title":null,"venue":null,"work_id":"39593abb-309a-403d-b771-c6252ccd5b07","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.017910Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:bc98fcc2c37805be1bf7d75fd921f0051816cf7f7136086c93018755fc2bdd98","observation_id":"6f4a3679-3a9f-4e9b-ad5f-e58ae87956b3","resolution":{"observed_at":"2026-08-12T00:37:59.658878Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.022697Z","title":"Representation Engineering: A Top-Down Approach to","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.022697Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:2bfe00dab9be4896c6fbab310a69014559b2fe315bc2c65bbafddd4ac26c3a87","observation_id":"0459a150-09f0-4ae5-a423-e6595df35361","resolution":{"observed_at":"2026-08-12T00:37:59.022697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.027822Z","title":"The Thirty-ninth Annual Conference on Neural Information Processing Systems , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.027822Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:7aecdd97ba56979083cd7634f71fd18d4cd5b9c2ede93359a5076aeb1e28263c","observation_id":"d2264ccc-03d7-408f-a9a2-2d9634c47017","resolution":{"observed_at":"2026-08-12T00:37:59.027822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.621929Z","title":"Llama Guard:","venue":null,"work_id":"5bb79503-4fec-4486-bb27-11186a50ba8a","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.032577Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:816124dd9e87e06ef3ad0bae5d6fa9b4caeefbc620088a8d3ca880ef0c51f209","observation_id":"b60c1275-addc-4a99-ad5e-ffeb853f940a","resolution":{"observed_at":"2026-08-12T00:37:59.626598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.607128Z","title":"ShieldGemma: Generative","venue":null,"work_id":"8ce60ee3-fd58-4368-8026-8b73c12d7065","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.037173Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:dbb2b6a456e11766daa66c46636a04841ebd1dafce43c51b393f920ed37bb34a","observation_id":"edb287ba-f08a-4853-a9ea-f7d7ca991962","resolution":{"observed_at":"2026-08-12T00:37:59.611291Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.592686Z","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of","venue":null,"work_id":"e7c7f5af-2016-4bf0-8521-4dcd78e16924","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.042273Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:b4352acbd4884fe0b70318fdd1a569c39cf3c5e25ac08c5f31b728eb1afe2b1a","observation_id":"089bc59a-3eb6-486d-952f-821c59431111","resolution":{"observed_at":"2026-08-12T00:37:59.597472Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.577610Z","title":"BeaverTails: Towards Improved Safety Alignment of","venue":null,"work_id":"9e8169d9-0348-478a-aa62-b7c7e216eba8","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.046851Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:bcf129efdf76d5c9f1d571367f083c7a49588ca524ab5d15c4769b579b2df1c9","observation_id":"403188b2-7140-4a34-97ce-fc980f85a667","resolution":{"observed_at":"2026-08-12T00:37:59.582454Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.562336Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"bbb597e2-38e3-4ea8-8e9f-58aa168b7e5d","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.051464Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:5815a900a4672b6f6374345685fc82b51af1ac49b96278a3801ab1f7c42a3566","observation_id":"37740555-f152-459d-bd26-6dcc8323aa5e","resolution":{"observed_at":"2026-08-12T00:37:59.567362Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.548165Z","title":null,"venue":null,"work_id":"a2a4939d-faa3-43ef-a04d-d154ab6c9cbd","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.055967Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:2553fdbbd4b584b0cd05675ef3f2e414d71187ba083c2958b30236888407492f","observation_id":"6bc35da8-d369-4766-9617-de5a12a58ab5","resolution":{"observed_at":"2026-08-12T00:37:59.552713Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.060666Z","title":"arXiv preprint arXiv:2601.04603 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.060666Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:ad8800c26d02acf84d1ebe98e035615cd299c8cc5b0d7610689162fc9e45c83a","observation_id":"e3fd81f6-42dd-4367-b6d4-2f5d23e04722","resolution":{"observed_at":"2026-08-12T00:37:59.060666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.065099Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.065099Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:6855ecf47f356354ce4e94ef4f1443735785675ae3ad90b851c1a72a70aeb908","observation_id":"fc24ee59-7394-4b51-be6f-63e0dacaacd6","resolution":{"observed_at":"2026-08-12T00:37:59.065099Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.069432Z","title":"International Conference on Learning Representations , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.069432Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:68e4f86f47ec77e77b14edd2308dc94df3534242a67ea465955ce3d211b9639b","observation_id":"25c05f1d-3c48-4ed7-aa4a-4348c8aa060e","resolution":{"observed_at":"2026-08-12T00:37:59.069432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.516211Z","title":"LLMScan: Causal Scan for","venue":null,"work_id":"93e07d92-ba75-4df2-a40b-e20db7009574","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.073680Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:aba1e28bb276518d21ba2d6256a1e1fdccf90c53a9bf73b477677f4a7056e7fc","observation_id":"2843ac01-abd3-4dd1-99b7-15d7a4d27312","resolution":{"observed_at":"2026-08-12T00:37:59.520545Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.501318Z","title":"Lightweight Safety Guardrails Using Fine-Tuned","venue":null,"work_id":"4c33dd80-e583-41c5-a4ca-cbdb227c0d92","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.077918Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:a4d5912d7635c79710aae83e4c904e2d590e2e354d87cb42c19133baf7dff719","observation_id":"634ecce7-7a5e-4470-b498-3b5aa80f3571","resolution":{"observed_at":"2026-08-12T00:37:59.506408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.082283Z","title":"Jailbroken: How Does","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.082283Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:e7e4afa9e0a5b6d69516b64b383c458f93ccd30a790ca5a66e616862cb1592d4","observation_id":"8dce6f83-8d76-4438-8959-87ecc1158e48","resolution":{"observed_at":"2026-08-12T00:37:59.082283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-08-12T09:06:50.363435Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-12T00:37:59.086570Z","title":"arXiv preprint arXiv:2307.15043 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.086570Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:5ae6c1f123d81c2b9ba8630d0a1a746a7182bcbd6f4fb8d9a26126e9e302cbfa","observation_id":"de493c2a-5478-498b-88df-56cf40d9b06b","resolution":{"observed_at":"2026-08-12T00:37:59.086570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14093","last_updated":"2024-12-20T02:22:19Z","snapshot_observed_at":"2026-08-13T11:12:58.507759Z","submitted_at":"2024-12-18T17:41:24Z","title":"Alignment faking in large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14093","snapshot_observed_at":"2026-08-12T00:37:59.091007Z","title":"arXiv preprint arXiv:2412.14093 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.091007Z"},"links":{"cited_paper":"/paper/2412.14093","citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:40941c9afccb9cfe2bc442c24a8dcd078ffa831427f1a8e09dbad903ca07db58","observation_id":"3dfcb42b-9a9b-4d04-b5c6-be648ef27ec9","resolution":{"observed_at":"2026-08-12T00:37:59.091007Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01822","last_updated":"2024-05-29T12:57:01Z","snapshot_observed_at":"2026-08-16T14:22:04.577037Z","submitted_at":"2024-02-02T16:35:00Z","title":"Building Guardrails for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01822","snapshot_observed_at":"2026-08-12T00:37:59.095718Z","title":"arXiv preprint arXiv:2402.01822 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.095718Z"},"links":{"cited_paper":"/paper/2402.01822","citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:50c55f935794dd172e0f332b716e0abca112c8276ea04c72ac2f9192f294d440","observation_id":"5ad8465c-c8be-45a3-a110-f651cb5e2067","resolution":{"observed_at":"2026-08-12T00:37:59.095718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.478358Z","title":"Attention Tracker: Detecting Prompt Injection Attacks in","venue":null,"work_id":"ce64330f-e8bd-442a-8e59-2b6486637fb0","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.099973Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:c7b264f77e026f423eed3b5d3e73c4fb9fe37647c97221955982fb1ea05e2222","observation_id":"28979a47-960d-4006-b226-afbcf44d1b69","resolution":{"observed_at":"2026-08-12T00:37:59.483031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.464495Z","title":"Probing Latent Subspaces in","venue":null,"work_id":"da2a5c47-bdd9-4c70-a996-795d9c83691b","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.103954Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:2f8c4dc2a1609ccf1ff768cfb7696e6168c543216c1d68320327d8c5a45fa1e1","observation_id":"bf60ba57-34d6-4a1d-8404-d53ac7ef3e24","resolution":{"observed_at":"2026-08-12T00:37:59.468493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.451240Z","title":"DeepContext: Stateful Real-Time Detection of Multi-Turn Adversarial Intent Drift in","venue":null,"work_id":"6553ec93-919a-4a75-8fd7-f34b6a84128c","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.107867Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:f7dcd6cf1f14e0be4ee2742aa47f85e308862928be60986593bf82e6ecd7b31a","observation_id":"cdae0327-410e-4279-9bee-7c0a8fd798be","resolution":{"observed_at":"2026-08-12T00:37:59.455445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.111778Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.111778Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:51573b49389ab9c7993b2841ba47c4e2b7746f02cdbc88c57d4fefb721d32d3e","observation_id":"f3acae53-8611-4c0c-a4b3-6fb77fa7b976","resolution":{"observed_at":"2026-08-12T00:37:59.111778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-08-17T11:08:48.802438Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-12T00:37:59.116162Z","title":"arXiv preprint arXiv:2407.10671 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.116162Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:17a6318fdaa5b9d44410d63b37a92ebe63278edc335bb982573fe42406c4293a","observation_id":"d7c658da-a5b0-4787-a0ec-ee021bffb0a6","resolution":{"observed_at":"2026-08-12T00:37:59.116162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00118","last_updated":"2024-10-02T15:22:49Z","snapshot_observed_at":"2026-08-02T16:20:09.773989Z","submitted_at":"2024-07-31T19:13:07Z","title":"Gemma 2: Improving Open Language Models at a Practical Size","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00118","snapshot_observed_at":"2026-08-12T00:37:59.121029Z","title":"arXiv preprint arXiv:2408.00118 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.121029Z"},"links":{"cited_paper":"/paper/2408.00118","citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:e20242fd16a03f0ae3dc0cefd115021b8e6f2d7b233ab6594852d8616aab17b6","observation_id":"f609ef15-9011-4242-ae66-c3678a3a0494","resolution":{"observed_at":"2026-08-12T00:37:59.121029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.426875Z","title":null,"venue":null,"work_id":"3bdc7cc6-464d-477f-b5ab-cb9dbbb7fa43","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.125213Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:dcd4086c9dab9906bcfaa6f726187130c76f670fcbd48b007be28c4981e0993c","observation_id":"14f4fe65-d58d-48d2-8113-6f11b288025c","resolution":{"observed_at":"2026-08-12T00:37:59.430852Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.411820Z","title":"Harmful Intent as a Geometrically Recoverable Feature of","venue":null,"work_id":"4b62154b-09c8-4d77-9989-e63361503f2b","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.129157Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:ecc0feb2f0071e04bfdd5e26ec9e3df79c0ca92e3718fcdb18f9f482fb835dfc","observation_id":"4f640a18-b274-4a2d-a025-cf64e22cb38f","resolution":{"observed_at":"2026-08-12T00:37:59.417130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12726","last_updated":"2026-07-09T09:24:42Z","snapshot_observed_at":"2026-08-16T05:54:24.149722Z","submitted_at":"2026-05-12T20:30:24Z","title":"Before the Last Token: Diagnosing Final-Token Safety Probe Failures","version":2},"cited_work":{"arxiv_id":"2605.12726","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.12726","snapshot_observed_at":"2026-08-12T00:37:59.179647Z","title":"Before the Last Token: Diagnosing Final-Token Safety Probe Failures","venue":"cs.LG","work_id":"66ea6432-640f-4a9c-b55b-5ec4319491f2","year":2026},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.133017Z"},"links":{"cited_paper":"/paper/2605.12726","citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:b8b12d327eb67728b0a85b25625848167f9c9ca0baaf68ad4f62d7ef17db641b","observation_id":"e4a529d2-ad31-4e7d-8d5d-ae1e3a121040","resolution":{"observed_at":"2026-08-12T00:37:59.186952Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.396937Z","title":null,"venue":null,"work_id":"03d3d479-0582-41c2-80bb-491f41a7b9ab","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.137247Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:97a40110a6ec5fffdad922db0b416e10715ea40f746ebe7923385f9204235c78","observation_id":"13053263-34da-4c37-9478-554028ac65da","resolution":{"observed_at":"2026-08-12T00:37:59.401677Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.381822Z","title":"ACM SIGOPS Operating Systems Review , volume=","venue":null,"work_id":"732f33ea-374a-4130-8316-beefa943e196","year":2025},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.141162Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:cc739c65e13220906f7df7fc50c4d3f0f92ae032a0c39999567c4c7c6b997d7f","observation_id":"036d2190-aa09-4e12-a31a-1f1d2a8aa5f9","resolution":{"observed_at":"2026-08-12T00:37:59.386755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T00:37:59.366701Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"8ce270a6-f46d-4385-90c7-b10d1ffa8a0c","year":null},"citing_paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-12T00:37:59.145162Z"},"links":{"citing_paper":"/paper/2608.08029"},"observation_digest":"sha256:296ba96988434e7ff67313da18fe1f746a3c8bbb30b710b2876e2e40e471774f","observation_id":"33784954-22a3-4fe7-b39e-cfa942e27e70","resolution":{"observed_at":"2026-08-12T00:37:59.371701Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2608.08029","last_updated":"2026-08-08T09:34:22Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-19T19:51:31.364367Z","submitted_at":"2026-08-08T09:34:22Z","title":"Do All LLMs Know When They're Being Harmful? A Reproducibility Study of Latent-Space Safety Probes Across Model Families"},"reference_resolution":{"displayed":31,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":16,"verified_exact":0,"verified_fuzzy":14},"total_outbound_references":31},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 20 August 2026, this Paper Citation Record lists 31 of 31 outbound references and 0 inbound Pith citation observations for arXiv:2608.08029."}