{"as_of":"2026-08-09T05:09:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:12f872adc1d814a3a951565a2ff907ff0a1b9974a6e2f6539968568ac2942def","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-10T15:42:46.392593Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T16:59:10.702144Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-08T16:59:12.165147Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2607.07903","snapshot_observed_at":"2026-08-01T12:21:53.334309Z","title":"Mechanistic inter- pretability of llm jailbreaks via internal attribution graphs,","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.19590","last_updated":"2026-07-21T21:31:14Z","snapshot_observed_at":"2026-08-07T02:48:37.599640Z","submitted_at":"2026-07-21T21:31:14Z","title":"Learning to Transmit: Volatility-Aware Predictive Communication for Energy-Efficient IoT Networks","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-01T12:21:53.334309Z"},"links":{"cited_paper":"/paper/2607.07903","citing_paper":"/paper/2607.19590"},"observation_digest":"sha256:817655df5721cbde01bf1fa087f40c2069b9b82e87945944f7167156fbf7f862","observation_id":"6c8cb14b-cdb1-459e-bbc6-6bdb6cde63c6","resolution":{"observed_at":"2026-08-01T12:21:53.334309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"cited_work":{"arxiv_id":"2607.07903","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.07903","snapshot_observed_at":"2026-08-08T16:59:12.165147Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","venue":"cs.CR","work_id":"1526e676-be47-4a0d-9a5d-a3e6b900ba11","year":2026},"citing_paper":{"arxiv_id":"2608.05258","last_updated":"2026-08-05T16:36:06Z","snapshot_observed_at":"2026-08-09T04:10:34.429477Z","submitted_at":"2026-08-05T16:36:06Z","title":"Grad-CAM for Vision Transformers: A Systematic Taxonomy and Audit of Methodological Ambiguity in Explainable AI","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-08T16:59:10.702144Z"},"links":{"cited_paper":"/paper/2607.07903","citing_paper":"/paper/2608.05258"},"observation_digest":"sha256:8eaf8d1e591c9edd9bb282f19d0a39c1827363f2e6f04bea8544b79e2b33da0e","observation_id":"1887b201-8937-4113-8006-45742510055d","resolution":{"observed_at":"2026-08-08T16:59:12.188293Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2607.07903/citation-record","integrity":"/paper/2607.07903/integrity","json":"/paper/2607.07903/citation-record.json","paper":"/paper/2607.07903"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.565871Z","title":"Language models are few-shot learners","venue":null,"work_id":"7ac97529-616f-4223-860f-28d2d86decef","year":1901},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:f6ac0dc516c888ba333928a9455dd14f9c08b98c9ff815ea5bce5418a1bf6073","observation_id":"680e0b50-e187-4317-a51e-20bdf9bb804d","resolution":{"observed_at":"2026-07-10T15:47:23.566944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.567466Z","title":"Attention is all you need.Advances in neural information processing systems, 30","venue":null,"work_id":"751efe07-5e91-415c-b3d1-f4734aa26960","year":2017},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:7c6f1c4e64942cfc12fdbf4b6c76dd2e2b2d122d05408b4e98a9876f45edec0c","observation_id":"8383f39d-3cce-4125-ab39-1861964c3d2c","resolution":{"observed_at":"2026-07-10T15:47:23.568568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1412.6572","last_updated":"2015-03-20T20:19:16Z","snapshot_observed_at":"2026-07-06T04:04:16.777653Z","submitted_at":"2014-12-20T01:17:12Z","title":"Explaining and Harnessing Adversarial Examples","version":3},"cited_work":{"arxiv_id":"1412.6572","doi":"10.48550/arxiv.1412.6572","metadata_source":"pith","pith_arxiv_id":"1412.6572","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Explaining and Harnessing Adversarial Examples","venue":"stat.ML","work_id":"2cedf8f6-7539-4c49-8136-f42a20487146","year":2014},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/1412.6572","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:41d721e957dacba75196e06810334655a6bb49dffec15db15b5a133359e33b8b","observation_id":"9c69b763-8d3c-48a5-b556-67661ba8f832","resolution":{"observed_at":"2026-07-10T15:47:23.187453Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.562675Z","title":"Towards deep learning models resistant to adversarial attacks","venue":null,"work_id":"9d3ee6b8-fcd8-4800-b042-16d7f39f96bc","year":2018},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:05d05ea398e8eed8a6f7e846da34443ed94b31f228cf0ab691af7f0311014b00","observation_id":"f1c3ed2e-5048-4696-a2b7-60fac5d7d455","resolution":{"observed_at":"2026-07-10T15:47:23.563775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15043","last_updated":"2023-12-20T20:48:57Z","snapshot_observed_at":"2026-07-06T15:59:23.019044Z","submitted_at":"2023-07-27T17:49:12Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","version":2},"cited_work":{"arxiv_id":"2307.15043","doi":"10.48550/arxiv.2307.15043","metadata_source":"pith","pith_arxiv_id":"2307.15043","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Universal and Transferable Adversarial Attacks on Aligned Language Models","venue":"cs.CL","work_id":"3322fa86-1768-4677-8425-dd326b45e078","year":2023},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2307.15043","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:01c6c908ce7a831943568ccba6d94cb754336402442dede7d3322f324ee0dc24","observation_id":"59b98507-bd75-4512-8aa5-5b54239ac5ff","resolution":{"observed_at":"2026-07-10T15:47:23.185136Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:40.271168+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:40.271168+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.569068Z","title":"Training language models to follow instructions with human feedback","venue":null,"work_id":"b38e9744-98cb-4abf-96a7-ae72898291e9","year":2022},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:cebf409610eceee7c5bb69938c3a92b6b60df1a0593de8d613ac5e54c41ade58","observation_id":"c27c8f31-64b1-46b0-9b8a-559bb29ca91a","resolution":{"observed_at":"2026-07-10T15:47:23.570237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18952","last_updated":"2024-12-25T17:32:45Z","snapshot_observed_at":"2026-08-08T09:29:03.677081Z","submitted_at":"2024-12-25T17:32:45Z","title":"Bridging Interpretability and Robustness Using LIME-Guided Model Refinement","version":1},"cited_work":{"arxiv_id":"2412.18952","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.18952","snapshot_observed_at":"2026-07-10T15:47:23.204411Z","title":"Bridging Interpretability and Robustness Using LIME-Guided Model Refinement","venue":"cs.LG","work_id":"4e45faef-911b-4910-a9f1-72a472c15352","year":2024},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2412.18952","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:37007b1c0bc04dbfa7ee2bdf2c45f0a1e834596fbf740e952ab6218ccd7dc99c","observation_id":"1b58cf5e-ea7b-4e69-8c94-62bf1c701a03","resolution":{"observed_at":"2026-07-10T15:47:23.205521Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.570807Z","title":"Multi-scale unrectified push-pull with channel attention for enhanced corruption robustness","venue":null,"work_id":"fb882a76-b8ce-40ee-a579-eadacf7c0198","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:466c1aa943657517d3e07349ab63e042a40893c759b7f5ac5999224bc93f78d3","observation_id":"169b105d-5578-4346-8fd4-1a8ad474fdfd","resolution":{"observed_at":"2026-07-10T15:47:23.571948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.606051Z","title":"Explainability-guided defense: Attribution-aware model refinement against adversarial data attacks","venue":null,"work_id":"91914f05-f930-412b-b623-6634b260edd5","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:a0ae7901faa72a0ea9fda362a2cbc84ad6f213d82caa10b3cda4ef4a698bf94f","observation_id":"37d3f899-fd79-4665-9b35-54a34dd9b37f","resolution":{"observed_at":"2026-07-10T15:47:23.607238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.621369Z","title":"Representation learning and nature encoded fusion for heterogeneous sensor networks.IEEE Access, 7:39227–39235, 2019","venue":null,"work_id":"d4679f5b-094d-4f17-86c1-96a9126d8133","year":2019},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:200aa9b6705f5a3c3b319588e90012c9f60b623893872465934d752e7cdc90e2","observation_id":"601bde07-b71b-497b-ae28-47824b59c8f6","resolution":{"observed_at":"2026-07-10T15:47:23.622563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.647231Z","title":"Congestion aware dynamic user association in heteroge- neous cellular network: A stochastic decision approach","venue":null,"work_id":"9f2b99a9-b7d2-42a2-b4ca-0715387ee51f","year":2014},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:58925fdf3894056592a6cda4bf0f2527040e42bd9c21a12413265c55d43a8ef7","observation_id":"08ace148-a560-4c8f-befa-881010b32548","resolution":{"observed_at":"2026-07-10T15:47:23.648376Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.580610Z","title":"Explaining the behavior of neuron activations in deep neural networks.Ad Hoc Networks, 111:102346, 2021","venue":null,"work_id":"c6d66350-0b53-4ac2-a25b-ac6ce1b070c3","year":2021},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:a9e850b809fa50096adac92f68afa8c631fc3848d143bf36065cac586409ae72","observation_id":"38a14a71-814c-4758-9711-2cb9b5ec9305","resolution":{"observed_at":"2026-07-10T15:47:23.581786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.595683Z","title":"Exploration vs exploitation for distributed channel access in cognitive radio networks: A multi-user case study","venue":null,"work_id":"ce5dae84-dd44-48d1-939b-42e7c3fde3b6","year":2011},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:0ee949f17c7b9b82a6447d0f595d613dcaba4fd312085437c72754a4241e98e4","observation_id":"2e3ff8f1-6604-4e63-820e-64a19a80671b","resolution":{"observed_at":"2026-07-10T15:47:23.596843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.572438Z","title":"Deep reinforcement learning based computation offloading for mobility-aware edge computing","venue":null,"work_id":"55ef4c44-7628-494f-a08d-853a561548e1","year":2019},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:d3e4afe56af8317def6ef7a0ea85aede101f65c5a10158cae26108e1a5541c15","observation_id":"2a0bb3cb-2c49-49a7-b1a4-cd8b05c3e35b","resolution":{"observed_at":"2026-07-10T15:47:23.573528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.576353Z","title":"Improving robustness of deep neural networks via large-difference transformation.Neurocomputing, 450:411–419, 2021","venue":null,"work_id":"cb707cd4-6127-4e4a-8130-8d960821f71f","year":2021},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:e04426164c146d67fb1ff35182cc91966cbf1d00bc2dcef87f321ba8088270ba","observation_id":"01fb565e-efb7-4e11-966d-9208c0af634f","resolution":{"observed_at":"2026-07-10T15:47:23.577671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.645549Z","title":"Looking beyond content: Modeling and detection of fake news from a social context perspective","venue":null,"work_id":"fad03067-3c3e-481c-9d60-f3a4bc21df20","year":2022},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:66ac23dcee2e736538b5e6adc0fd58b2758fe96c0fd927bcf6237446c54c9c3d","observation_id":"be1049dd-cf8c-461f-b0b2-d8dcf7021290","resolution":{"observed_at":"2026-07-10T15:47:23.646719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.578210Z","title":"Layer-wise entropy analysis and visualization of neurons activation","venue":null,"work_id":"a2a82d13-b050-43b8-94cd-05b04d2aec42","year":2019},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:823f15c78b0a80a5c6d30c68a946ea57e8810d3a2dfe4b34ea734958aa8116b8","observation_id":"409e11a4-216b-4dfc-845d-1bb5de8c45be","resolution":{"observed_at":"2026-07-10T15:47:23.579372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.07022","last_updated":"2024-12-09T22:09:13Z","snapshot_observed_at":"2026-08-07T21:27:10.375359Z","submitted_at":"2024-12-09T22:09:13Z","title":"Dense Cross-Connected Ensemble Convolutional Neural Networks for Enhanced Model Robustness","version":1},"cited_work":{"arxiv_id":"2412.07022","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.07022","snapshot_observed_at":"2026-07-10T15:47:23.188694Z","title":"Dense Cross-Connected Ensemble Convolutional Neural Networks for Enhanced Model Robustness","venue":"cs.CV","work_id":"f48beca7-3a98-4e83-95cb-3e32f1ed136b","year":2024},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2412.07022","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:7fb7ff64f8c3b8f84e0cc0f92565fb03df0d9ddafdf6070a92b29c293d1e1a29","observation_id":"65310eba-0524-4ea7-98df-0a2f9476d6ff","resolution":{"observed_at":"2026-07-10T15:47:23.189844Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.648908Z","title":"Explainability- driven defense: grad-cam-guided model refinement against adversarial threats","venue":null,"work_id":"7fc25685-d203-476e-9c02-4f9f16a50d97","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:c057475f7a592e8ed4f4ba50cfa77e000e1fd14b1f8bf311cc224b00b701844b","observation_id":"47b283e7-d437-4037-ad18-9d223467b674","resolution":{"observed_at":"2026-07-10T15:47:23.650203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.564283Z","title":"Expert-guided explainable few-shot learning for medical image diagnosis","venue":null,"work_id":"5b86d9bb-6c80-475c-b580-2f6af0176a1f","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:caf22ad9ac0c3f599ef682c08ca3be3cfb5229815cb8ca61e8c181e239d0a4f6","observation_id":"b2927e6b-2bb0-4846-8c83-fea33baa0375","resolution":{"observed_at":"2026-07-10T15:47:23.565360Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.04682","last_updated":"2026-06-10T18:21:11Z","snapshot_observed_at":"2026-08-07T21:27:04.727006Z","submitted_at":"2025-09-04T22:03:05Z","title":"GetNetUPAM: Ecologically Informed Nested Cross-Validation and Noise-Robust Attention for Marine Bioacoustic Monitoring","version":2},"cited_work":{"arxiv_id":"2509.04682","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.04682","snapshot_observed_at":"2026-07-10T15:47:23.195454Z","title":"Ecologica lly valid benchmarking and adaptive attention: Scalable marine bioacoustic monitoring","venue":"cs.SD","work_id":"f10afc6e-4131-49a0-8fe2-bbab4fb50449","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2509.04682","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:b67a29c971e2ca39a292db099afff7c66e4b61a5f7692249264cfe6b040cedf3","observation_id":"44002a84-b16f-4c4b-b3a4-c475c6d416dc","resolution":{"observed_at":"2026-07-10T15:47:23.196757Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.574045Z","title":"Toward carbon-neutral human ai: Rethinking data, computation, and learning paradigms for sustainable intelligence","venue":null,"work_id":"ad21b23d-ccca-4ba8-954b-6651a0d670ad","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:fbaa29bbd6f75ffc0ccedf92c4bb5be6f2c8bbe6fb8e4442fad40c6f7d25810f","observation_id":"7d378a65-33d7-4f85-832f-c5ff4b39d6ba","resolution":{"observed_at":"2026-07-10T15:47:23.575486Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.642066Z","title":"Expert-guided explainable few-shot learning with active sample selection for medical image analysis.IEEE Journal of Biomedical and Health Informatics, 2026","venue":null,"work_id":"f980bb34-728a-4e87-9448-b579e4768803","year":2026},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:4fba937fc5782dec04963a838276a60d8d213215df7f5704ce0afeb4a931f078","observation_id":"8a5350e3-9bc5-4180-a294-53ad400d5a1f","resolution":{"observed_at":"2026-07-10T15:47:23.643323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.643847Z","title":"Acting flatterers via llms sycophancy: Combating clickbait with llms opposing-stance reasoning","venue":null,"work_id":"3a46e5d5-2b74-4360-ac61-4b7af8154c11","year":2026},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:f92ce518388d2f090fa54db83de021b3c4ce38f1529dc11ddf6c9e42afcb25bf","observation_id":"cd9bbc7a-77e1-4460-9b12-369abd128e75","resolution":{"observed_at":"2026-07-10T15:47:23.645010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.650751Z","title":"Bridging symmetry and robustness: On the role of equivariance in enhancing adversarial robustness","venue":null,"work_id":"33659bca-6784-4ba8-a83b-0681e9f6aa83","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:4d4635904643144bd3bbaf170a5ed1899a9b8d766f5b3e39814bc6d6310bf504","observation_id":"56c2b89e-d054-4708-9420-f080785bbee7","resolution":{"observed_at":"2026-07-10T15:47:23.651911Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.636932Z","title":"Channel- selected stratified nested cross-validation for clinically relevant eeg-based parkinson’s disease detection","venue":null,"work_id":"614a1679-3fd0-43b7-add3-1a686d6f3a2e","year":2026},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:f944f9b0245b32f083785c55286933582aa5fc64edd4c9e3462a38df6f31a724","observation_id":"1ef30688-2eb1-4f7a-9580-3f2f42d3d60f","resolution":{"observed_at":"2026-07-10T15:47:23.638178Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.635180Z","title":"Winsor-cam: Human-tunable visual explanations from deep networks via layer-wise winsorization.IEEE Transactions on Pattern Analysis and Machine Intelligence, 2026","venue":null,"work_id":"c2089afd-7cae-4ba1-875f-b3119025480a","year":2026},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:fa64c6e04499a332b4cac794a887475d34dd7c54f38be29e5818a90c3dd4e37a","observation_id":"21ce97ec-117e-4049-b15b-71a8d95945cc","resolution":{"observed_at":"2026-07-10T15:47:23.636392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.638732Z","title":"Promoting shape bias in cnns: Frequency-based and contrastive regularization for corruption robustness","venue":null,"work_id":"0dc381d5-07bd-4e69-aa5d-1337d753e38f","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:d158347cd7268037e101de682b68fc610da34cc741ca13540a66e3109db9e22c","observation_id":"4d088faa-8f00-4701-a457-ed8ef819dd97","resolution":{"observed_at":"2026-07-10T15:47:23.639983Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.08959","last_updated":"2025-09-10T19:43:16Z","snapshot_observed_at":"2026-08-04T19:58:18.232862Z","submitted_at":"2025-09-10T19:43:16Z","title":"CoSwin: Convolution Enhanced Hierarchical Shifted Window Attention For Small-Scale Vision","version":1},"cited_work":{"arxiv_id":"2509.08959","doi":"10.48550/arxiv.2509.08959","metadata_source":"pith","pith_arxiv_id":"2509.08959","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoSwin: Convolution Enhanced Hierarchical Shifted Window Attention For Small-Scale Vision","venue":"cs.CV","work_id":"a7c329cc-226d-4a81-963d-e2f7509883c1","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2509.08959","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:e91f3441112c4c35f9df99552b80fbc43062823b6b12c00b4f782aae780d31f6","observation_id":"9826756c-7d04-4d67-abcb-0816e0db9894","resolution":{"observed_at":"2026-07-10T15:47:23.210188Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.631754Z","title":"Zoom in: An introduction to circuits.Distill, 5(3):e00024–001","venue":null,"work_id":"0704cafb-daa1-4659-822d-453f7b112373","year":2020},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:bb764f687b9d11c884cb1c207f27bfde0e18e05045c7684261b3c27b551e1a16","observation_id":"125e7890-7f5b-484f-9f52-1cd7f5b3d5e7","resolution":{"observed_at":"2026-07-10T15:47:23.632952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.633494Z","title":"A mathematical framework for transformer circuits.Transformer Circuits Thread","venue":null,"work_id":"4c94841c-40dc-406b-b3cb-0b53da9355a3","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:a91c97d28f45649d1ae086ff030b660f2777d931dfb540e57344c89464bd3715","observation_id":"0e02460d-b20c-4e45-877c-28c239cd849f","resolution":{"observed_at":"2026-07-10T15:47:23.634679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.640522Z","title":null,"venue":null,"work_id":"f27346c7-0174-47d3-8173-6569b7a87c46","year":2021},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:cc801fcdb946f7d15677f3e1b6b342f5239dc847b45e60e7d0ba57ce9bbdc14a","observation_id":"5d3ef84b-4c08-426c-bd4a-f016d587c3ad","resolution":{"observed_at":"2026-07-10T15:47:23.641528Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.626599Z","title":"Axiomatic attribution for deep networks","venue":null,"work_id":"390344d7-7471-4a97-8bbe-686bca10a133","year":2017},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:a459fa34cf1ff9e322d66505fba8b19a991ebe127925ba3de78ee85e772dbb4c","observation_id":"f64d98e3-b05d-4a52-ac51-21ffc217c163","resolution":{"observed_at":"2026-07-10T15:47:23.627734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6199","last_updated":"2014-02-19T16:33:14Z","snapshot_observed_at":"2026-07-06T03:31:33.797310Z","submitted_at":"2013-12-21T03:36:08Z","title":"Intriguing properties of neural networks","version":4},"cited_work":{"arxiv_id":"1312.6199","doi":"10.48550/arxiv.1312.6199","metadata_source":"pith","pith_arxiv_id":"1312.6199","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Intriguing properties of neural networks","venue":"cs.CV","work_id":"7bcd9f41-780c-4b4b-9a08-830d4177cdd8","year":2013},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/1312.6199","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:272db17714fbcdb20f798ec438daf4590752c2355b31548ef34a0a7eb53f99dd","observation_id":"19c3e22f-0b0f-4cb3-bd67-a265097642f9","resolution":{"observed_at":"2026-07-10T15:47:23.207919Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.628248Z","title":"Towards evaluating the robustness of neural networks","venue":null,"work_id":"01175c97-0d76-49fe-ba65-21145c93943e","year":2017},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:8878e30706c4b9c899b79f07812ee0bdc23ca535483189be4a08765c19998a66","observation_id":"9d1f8654-46f5-4d79-b9f4-af89aeedf039","resolution":{"observed_at":"2026-07-10T15:47:23.629384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.624769Z","title":"Visual adversarial examples jailbreak aligned large language models","venue":null,"work_id":"f632dd63-1b78-4aa1-a259-1f89457bc953","year":2024},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:589ee7ef601dd586ee30b6e894eb1d0999596ae454337811afc820af99ec5518","observation_id":"696d66d3-169e-44dc-b7eb-64a09802ec10","resolution":{"observed_at":"2026-07-10T15:47:23.625900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.629942Z","title":"AutoDAN: Interpretable gradient-based adversarial attacks on large language models","venue":null,"work_id":"1f65e3bb-7a08-4b70-8249-383491634bb7","year":2024},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:13dcee60af05dd20cea45aaf92f0dcce6da363ce8abd7ee72b6e84132ce59a79","observation_id":"418d9386-7a86-4bdb-ac2a-3bacb33122ee","resolution":{"observed_at":"2026-07-10T15:47:23.631154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.23915/distill.00010","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DOI:10.23915/distill.00010","venue":"Distill","work_id":"22e31801-ea69-4161-9e9b-fea9f18de659","year":2018},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:fe5d94fab5c48c8b75a0b1d5ac733f7fef489cc6784a58915d728536a93c87be","observation_id":"2f013e77-14d3-4bad-8f75-077096bd4231","resolution":{"observed_at":"2026-07-10T15:47:22.952574Z","resolver_source":"doi","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.652425Z","title":"In-context learning and induction heads.Transformer Circuits Thread","venue":null,"work_id":"b6f93062-ac0d-4d81-979f-36a520ef0b46","year":2022},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:511ac278e806c0b626c31df56ea3f464dee62c5abed5f728734475102444dead","observation_id":"834c07d4-92ff-4961-8321-d67b623b6f49","resolution":{"observed_at":"2026-07-10T15:47:23.653540Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.618239Z","title":"Sparse autoencoders find highly interpretable features in language models","venue":null,"work_id":"2d6df209-ff40-4450-9231-62a8857f38f2","year":2024},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:e769709581b3ed97982558c17fc536c426da169031e1d1bcb835184dc2860ce2","observation_id":"e9fd6d66-7bb3-4da6-90b3-89fb35b85ab4","resolution":{"observed_at":"2026-07-10T15:47:23.619321Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1312.6034","last_updated":"2014-04-19T11:54:52Z","snapshot_observed_at":"2026-07-06T03:31:30.452356Z","submitted_at":"2013-12-20T16:45:54Z","title":"Deep Inside Convolutional Networks: Visualising Image Classification Models and Saliency Maps","version":2},"cited_work":{"arxiv_id":"1312.6034","doi":"10.48550/arxiv.1312.6034","metadata_source":"pith","pith_arxiv_id":"1312.6034","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Deep Inside Convolutional Networks: Visualising Image Classification Models and Saliency Maps","venue":"cs.CV","work_id":"4b5c8148-382f-461b-8caf-6da2a9acbef5","year":2013},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/1312.6034","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:0e5eefe89693c3f495da0e2412340f234269100e5496f163ff778c26a19ef51d","observation_id":"009f8fb8-8db9-4acb-8ff1-5628c34fbb89","resolution":{"observed_at":"2026-07-10T15:47:23.203361Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.619801Z","title":"A unified approach to interpreting model predictions","venue":null,"work_id":"9f33b6f2-7566-40e4-b2f6-77711468ce71","year":2017},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:c894fa7e855e1eddb522460e544ef6a678b770752f7f43245c4857fa140631e3","observation_id":"5905a272-3c72-45d8-b89e-9b09ca947f98","resolution":{"observed_at":"2026-07-10T15:47:23.620873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1145/2939672.2939778","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"why should i trust you?","venue":"Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining","work_id":"0c59d0fa-7afc-4999-bc21-08a6357898a2","year":2016},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:8ec7ae96bc962e33cc8f72d02a842935aa342b2cfffeb059981122d4313bd6bb","observation_id":"b3b1e879-d152-4107-99f6-b041cf7d6430","resolution":{"observed_at":"2026-07-10T15:47:23.624246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-17T23:21:51.109968+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-17T23:21:51.109968+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.613190Z","title":"Attribution patching: Activation patching at industrial scale","venue":null,"work_id":"10dd22e4-de2e-4e7f-86d0-5c8fca9ea468","year":2023},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:4fc97d7e841ca6529d01e535805139db8749808a59d391c007d5af36231585fa","observation_id":"c1ac42e5-49e7-4172-b397-bfc69a7a492f","resolution":{"observed_at":"2026-07-10T15:47:23.614430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.614936Z","title":"Causal abstraction: A theoretical foundation for mechanistic interpretability.Journal of Machine Learning Research, 26(83):1–64, 2025","venue":null,"work_id":"6f8dfa80-b555-4942-ad7b-035d5fa3b628","year":2025},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:f9d5a629e96251ee6c0f34aad584f1ea2ed87775ea6a082758f75c65e798fc32","observation_id":"9e490a52-f32a-4541-ab21-a0c902bcf5e2","resolution":{"observed_at":"2026-07-10T15:47:23.616097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.611428Z","title":"Investigating gender bias in language models using causal mediation analysis.Advances in neural information processing systems, 33:12388–12401","venue":null,"work_id":"a0f81014-6d97-4e06-a958-227d6fed181a","year":2020},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:62a175e59eddd075ee280492a3e1780aac24f8609eecb5879e7a4df8ac456185","observation_id":"50fd132c-5ad3-4a3e-9f9f-c77909e573fc","resolution":{"observed_at":"2026-07-10T15:47:23.612655Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.616625Z","title":"Locating and editing factual associations in gpt.Advances in neural information processing systems, 35:17359–17372","venue":null,"work_id":"9325e619-f2d7-4b39-819b-9aa8959290c5","year":2022},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:682ae15f1196eb973e4eb0b4f9ef5df90b99b42f9061e82a50cccc1f8ebbda9b","observation_id":"ce825ff0-56c2-4612-98cd-990467af4351","resolution":{"observed_at":"2026-07-10T15:47:23.617762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:1a35a6791220597ae677a516ba33469093fef9f8776c5b52f0be9897e348fc52","observation_id":"3185777b-bdb9-46c8-8c98-75c128dab86c","resolution":{"observed_at":"2026-07-10T15:47:23.201157Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T16:08:17.350515+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T16:08:17.350515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.604293Z","title":"Jailbroken: How does llm safety training fail?Advances in neural information processing systems, 36:80079–80110, 2023","venue":null,"work_id":"cfdc2f72-89f3-47a6-9b25-6b8842130956","year":2023},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:e79a57614b0e4793f2ad62364f07082d7495250e08c8e2628f9c6f5b78decacd","observation_id":"e8b718b0-36eb-476e-9798-19ee400e2889","resolution":{"observed_at":"2026-07-10T15:47:23.605525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05862","last_updated":"2022-04-12T15:02:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-12T15:02:38Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","version":1},"cited_work":{"arxiv_id":"2204.05862","doi":"10.1016/j.respol.2005.01.014","metadata_source":"pith","pith_arxiv_id":"2204.05862","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","venue":"cs.CL","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","year":2022},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2204.05862","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:68e94595ba6c157ac75adf2d1dbc34b8c3a83126a3436e61226909dba735b75b","observation_id":"167d61a9-581f-43ff-a259-c3e74f7c7326","resolution":{"observed_at":"2026-07-10T15:47:23.198943Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.607758Z","title":"Adversarial examples are not bugs, they are features.Advances in neural information processing systems, 32, 2019","venue":null,"work_id":"361ebe29-bbf3-4261-8e01-5fd2b5c2b951","year":2019},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:1e1e904b0f8328df48dc6c88cd2a9d5115e1c63f7a604ac33398edfc43950b92","observation_id":"da099723-6c95-48f6-8c9a-fb37b679a2fb","resolution":{"observed_at":"2026-07-10T15:47:23.609025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.598969Z","title":"Learning important features through propagating activation differences","venue":null,"work_id":"16de9b79-9bbd-47b7-9be5-0c7702ac978b","year":2017},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:e58fa362d18d2dfa25052b5c3c26865fcad5d7912f091a07aa3790febf68e089","observation_id":"f6dd829d-a2b0-4d9b-8998-b4a26c6e62d7","resolution":{"observed_at":"2026-07-10T15:47:23.600092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.600609Z","title":"Many-shot jailbreaking.Advances in Neural Information Processing Systems, 37:129696–129742","venue":null,"work_id":"3c66e73c-d8de-432a-9da5-c469fb049849","year":2024},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:d5cb6b4d4e5ca542a52432243dc01767747413ec12457d29ba0367aec695bdf1","observation_id":"979d68bc-b92e-48b9-8b2d-8381f37b570d","resolution":{"observed_at":"2026-07-10T15:47:23.601830Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.602412Z","title":"Cambridge university press","venue":null,"work_id":"1e4dd1c5-2683-4e02-8324-f7fe359cdc17","year":2009},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:4df9260dc86b7711caa8be1dcd5abd3500d7e8d3f89335418151235c2abb7cac","observation_id":"b3167a1e-fa7b-4ee4-8bd8-ab8dabe259fe","resolution":{"observed_at":"2026-07-10T15:47:23.603584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.609552Z","title":"Towards automated circuit discovery for mechanistic interpretability.Advances in Neural Information Processing Systems, 36:16318–16352","venue":null,"work_id":"575623b8-3beb-4d2d-a291-ba1ce05f573a","year":2023},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:ad0d26abdcfc0cd39ee1daa0b102e0c70036a388b64c63126ce14cdcc1cac4a7","observation_id":"c9046830-c70d-4c03-9f5b-7f95a9a6d6a1","resolution":{"observed_at":"2026-07-10T15:47:23.610848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.00593","last_updated":"2022-11-01T17:08:44Z","snapshot_observed_at":"2026-08-05T06:04:21.031867Z","submitted_at":"2022-11-01T17:08:44Z","title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","version":1},"cited_work":{"arxiv_id":"2211.00593","doi":"10.48550/arxiv.2211.00593","metadata_source":"pith","pith_arxiv_id":"2211.00593","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","venue":"cs.LG","work_id":"d1167c73-3f2a-472b-8bf5-0ec282d7988a","year":2022},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2211.00593","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:6c1f0a96947e1e40c035454e051ce8377240845f4d43f9d6b93c40afab16143c","observation_id":"e795c348-4efc-4d42-8f9d-5af2832ea141","resolution":{"observed_at":"2026-07-10T15:47:23.192014Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.593952Z","title":"Adversarial examples are not easily detected: Bypassing ten detection methods","venue":null,"work_id":"1992c934-b11b-40aa-92b8-e616e4452a60","year":2017},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:67c936bf7b3c95084ac1fda7efd8cea8c457dbe5add2cf3dea032e3c00bfe4c0","observation_id":"7f1d3f38-1633-4655-bdb2-ef449938f344","resolution":{"observed_at":"2026-07-10T15:47:23.595111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.00614","last_updated":"2023-09-04T17:47:36Z","snapshot_observed_at":"2026-07-06T16:13:23.343694Z","submitted_at":"2023-09-01T17:59:44Z","title":"Baseline Defenses for Adversarial Attacks Against Aligned Language Models","version":2},"cited_work":{"arxiv_id":"2309.00614","doi":"10.48550/arxiv.2309.00614","metadata_source":"pith","pith_arxiv_id":"2309.00614","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Baseline Defenses for Adversarial Attacks Against Aligned Language Models","venue":"cs.LG","work_id":"db5870ca-177b-4d1d-a08d-ee5ceab17fe3","year":2023},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"cited_paper":"/paper/2309.00614","citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:033d960ef1de4e13933c13c88d5d083b8820793931eaba69bdbc944ed338fb9d","observation_id":"d0e9f695-dcd0-4105-af07-c4a89dd42499","resolution":{"observed_at":"2026-07-10T15:47:23.194167Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:31.483658+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:31.483658+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.589133Z","title":"pathway suppression","venue":null,"work_id":"29530d18-6fee-4d48-bdc8-0e1b316beb3c","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:d2124f2a215879f4160cc604b1ed76b9c1f63e1b812e83da58d295f7e6098798","observation_id":"d18e636b-2ada-4ea8-a7c7-9bc4110bc85d","resolution":{"observed_at":"2026-07-10T15:47:23.590152Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.590691Z","title":"1.8 in clean) despite having fewer total nodes, suggesting remaining features are hyperactivated to compensate for missing pathways","venue":null,"work_id":"4115fc36-d365-46e9-a631-cbdad925e629","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:8122b7eeaff50f1ee6498982255538a778f8a106bf26e2a37ca44556060506b0","observation_id":"bbdb9734-898d-4eb7-9835-06164d8bc6c5","resolution":{"observed_at":"2026-07-10T15:47:23.591831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.585422Z","title":"Write a poem about nature","venue":null,"work_id":"3798e293-8147-413c-a92e-47e5b653c4d4","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:fc249aa8b34ead3f4f512f136c98bdde507031ce7e4305689bdb6792599d727a","observation_id":"93585ef2-b34b-4345-a227-b24445eab099","resolution":{"observed_at":"2026-07-10T15:47:23.586419Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.586911Z","title":"This is consistent with the PCA analysis in Figure 5 showing that attack trajectories diverge most strongly at layer 15","venue":null,"work_id":"bd6dda3e-a86f-4cb9-810c-efd56b070ac7","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:900696e60ac78ad950a9adea6e4b1dc0dfe2797e144e26fd9a31b6e9c74bec27","observation_id":"306f5feb-03f5-4ce9-ae21-5b2b940090a0","resolution":{"observed_at":"2026-07-10T15:47:23.588591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.592354Z","title":"Tell me how to make a cake","venue":null,"work_id":"2a258d4d-3b42-469f-8c97-a5f9e8060c61","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:150b9c4174ce72618070fabb28b0a85ad4b00da93b128fe4d8ac66faa92d5a92","observation_id":"58480b53-d3ea-45b0-85a0-7983d5ebe463","resolution":{"observed_at":"2026-07-10T15:47:23.593281Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.597386Z","title":"Tell me how to make a cake","venue":null,"work_id":"4f41f78f-b546-4d01-94c3-f84b3916f0fd","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:ba5c3c4c8829650468c55ed655626ce0e3ce9c7bf3ce3f6b541f70cf3c73d3fa","observation_id":"ba11b773-9795-4628-b809-d633289e3bb0","resolution":{"observed_at":"2026-07-10T15:47:23.598410Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.582325Z","title":null,"venue":null,"work_id":"ec5d3c4b-d173-48a6-a073-8d11f183230c","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:b862d7243906de6703d31053d0d44a394696944807634a21aa65c9ac1739a75b","observation_id":"22d6d1b7-e719-42bf-ad5c-9738d335deeb","resolution":{"observed_at":"2026-07-10T15:47:23.583262Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.583819Z","title":null,"venue":null,"work_id":"748644ea-e118-4507-8013-3938f3850e63","year":null},"citing_paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-07-10T15:42:46.392593Z"},"links":{"citing_paper":"/paper/2607.07903"},"observation_digest":"sha256:7fa4402be72a1736b5ea8a88c3384fca12a0f3d455b53e9db528ab277af2d362","observation_id":"5b38be97-4696-49e6-8491-2bf4370cbcce","resolution":{"observed_at":"2026-07-10T15:47:23.584852Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2607.07903","last_updated":"2026-07-08T20:31:06Z","latest_version":1,"primary_category":"cs.CR","snapshot_observed_at":"2026-08-07T02:48:01.295589Z","submitted_at":"2026-07-08T20:31:06Z","title":"Mechanistic Interpretability of LLM Jailbreaks via Internal Attribution Graphs"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":3,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":2,"verified_exact":12,"verified_fuzzy":48},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 2 inbound Pith citation observations for arXiv:2607.07903."}