{"as_of":"2026-08-17T13:24:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3d7d61b32437691079c3a875b055ee0f0ed0fdaa520e8da5960c0b8826931f83","coverage":[{"denominator":105,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T04:25:06.108473Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-12T02:42:26.173782Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"cited_work":{"arxiv_id":"2505.01372","doi":"10.48550/arxiv.2505.01372","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.01372","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating","venue":"ArXiv.org","work_id":"f4a6a168-76c4-4b24-8d89-8609af3aa4e7","year":2025},"citing_paper":{"arxiv_id":"2605.08934","last_updated":"2026-06-16T20:17:16Z","snapshot_observed_at":"2026-08-11T08:05:03.537052Z","submitted_at":"2026-05-09T13:08:07Z","title":"From Mechanistic to Compositional Interpretability","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-12T02:42:26.173782Z"},"links":{"cited_paper":"/paper/2505.01372","citing_paper":"/paper/2605.08934"},"observation_digest":"sha256:ce347c5f26db7fb1970b31d609022778642917dfc0537f17c77b7f3c2aa5bf40","observation_id":"c7a68c25-a842-4499-bca8-0cea977ccc1b","resolution":{"observed_at":"2026-05-12T02:46:18.444788Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.01372/citation-record","integrity":"/paper/2505.01372/integrity","json":"/paper/2505.01372/citation-record.json","paper":"/paper/2505.01372"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.08025","last_updated":"2025-04-01T14:16:47Z","snapshot_observed_at":"2026-08-17T08:00:35.522399Z","submitted_at":"2024-10-10T15:22:48Z","title":"The Computational Complexity of Circuit Discovery for Inner Interpretability","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.08025","snapshot_observed_at":"2026-08-16T04:25:05.518727Z","title":"The computational complexity of circuit discovery for inner interpretability","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.518727Z"},"links":{"cited_paper":"/paper/2410.08025","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d04d8f7c36a6fca0cda02aa535e816d4400f07379a018fe0b11c2ce4eb74df81","observation_id":"7be2a119-13da-4a7a-a1d7-649e3984d9d1","resolution":{"observed_at":"2026-08-16T04:25:05.518727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14316","last_updated":"2024-07-16T10:22:51Z","snapshot_observed_at":"2026-08-16T14:57:46.371369Z","submitted_at":"2023-09-25T17:37:20Z","title":"Physics of Language Models: Part 3.1, Knowledge Storage and Extraction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14316","snapshot_observed_at":"2026-08-16T04:25:05.526366Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.526366Z"},"links":{"cited_paper":"/paper/2309.14316","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:e0f5cf40654d6854673a48450c53aa7cd4dada68f02e4eb2ec4ca1b1047e78af","observation_id":"e96be6eb-3233-415f-bede-dba250dc41a7","resolution":{"observed_at":"2026-08-16T04:25:05.526366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.531896Z","title":"Physics of language models: Part 3.1, knowledge storage and extraction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.531896Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:4760219a2d4263634375796dec7e18d7e83eda94e795a752b0e8230bab1ca6ca","observation_id":"61d387c8-950c-4507-82b2-d3eb05f6a283","resolution":{"observed_at":"2026-08-16T04:25:05.531896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.538644Z","title":"The urgency of interpretability, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.538644Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:36509869b46cd96364b16f52d12da11be94e109af5d334b8d5f2d4e4a534a500","observation_id":"aae00ee6-8ee0-419b-adff-b23b52bf598b","resolution":{"observed_at":"2026-08-16T04:25:05.538644Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09932","last_updated":"2024-09-06T00:46:40Z","snapshot_observed_at":"2026-08-16T14:00:45.842560Z","submitted_at":"2024-04-15T16:58:28Z","title":"Foundational Challenges in Assuring Alignment and Safety of Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09932","snapshot_observed_at":"2026-08-16T04:25:05.544723Z","title":"Foundational challenges in assuring alignment and safety of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.544723Z"},"links":{"cited_paper":"/paper/2404.09932","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f52fc5a0c009dc6c1487dac83fc44d0fa423fec9b3ce8606971ce6cac8c7efd2","observation_id":"deff4b33-b745-4e3a-9e7a-86e1815bb557","resolution":{"observed_at":"2026-08-16T04:25:05.544723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.550537Z","title":"Ai as systems, not just models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.550537Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6fc133a89046c559122a0a3a927d329a8feb43ff5ef10b381cc6ea1e99f89507","observation_id":"964da59b-029d-4588-8631-c9b228d56532","resolution":{"observed_at":"2026-08-16T04:25:05.550537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.556998Z","title":"Standard saes might be incoherent: A choosing problem & a “concise” solution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.556998Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c4c4775005c4726d57d3f9c42da1f65f5fede38f17336c893647fb12d235262d","observation_id":"7c179344-ca3a-48d8-82d1-3ba93e274d41","resolution":{"observed_at":"2026-08-16T04:25:05.556998Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.563027Z","title":"Position: Interpretability is a bidirectional communication problem","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.563027Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:8027f7f4ee661ba4db3043a2209f3660bd345b3fe7b36d05ad6209748266bec7","observation_id":"1000c22c-0b9a-4e30-8e2f-592afc155ec8","resolution":{"observed_at":"2026-08-16T04:25:05.563027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.569755Z","title":"A mathematical philosophy of explanations in mechanistic interpretability: The strange science part i.i, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.569755Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:de2bb42f7baa932a441fd3ac8dea03af2d8e852271f7c24825b3d4d0da1064e5","observation_id":"237b7281-268c-41a9-b03e-66dee8e8639d","resolution":{"observed_at":"2026-08-16T04:25:05.569755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11179","last_updated":"2024-10-15T01:38:03Z","snapshot_observed_at":"2026-08-16T13:09:22.337092Z","submitted_at":"2024-10-15T01:38:03Z","title":"Interpretability as Compression: Reconsidering SAE Explanations of Neural Activations with MDL-SAEs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.11179","snapshot_observed_at":"2026-08-16T04:25:05.575857Z","title":"Pearce, and Lee Sharkey","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.575857Z"},"links":{"cited_paper":"/paper/2410.11179","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0145797dfacda381e863bfa32b99fcf9af47c1be985938125fdfcc14740c3633","observation_id":"2f798255-8dd9-4727-8730-208b6e6b64e5","resolution":{"observed_at":"2026-08-16T04:25:05.575857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.581703Z","title":"Novum Organum","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.581703Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:76ba08fb4661bcf5719644f3b7b16b6438107e82b46e7cd852b82bb3f2d75658","observation_id":"e326204e-4a5b-4139-967e-9d642129ebdf","resolution":{"observed_at":"2026-08-16T04:25:05.581703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.586827Z","title":"Simplicity","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.586827Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7fbd941b8ec561f17fa0f64583093cd482669a7da6a7510d5a15af9d4eddc15e","observation_id":"221e6ee1-3828-4e38-ba66-1b3543b46e24","resolution":{"observed_at":"2026-08-16T04:25:05.586827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.592337Z","title":"Design Rules: The Power of Modularity Volume 1","venue":null,"work_id":null,"year":1999},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.592337Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0ff85f186b6cc0eb2230b0089dcc640300a53e8415df0d73acd90e497f3e6e3e","observation_id":"becd414b-4fff-4ff3-954f-598d5526b840","resolution":{"observed_at":"2026-08-16T04:25:05.592337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.597081Z","title":"Icml 2024 mechanistic interpretability workshop, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.597081Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:fb23970da7469acfbcc9f00db9f543d55c5e30e0175f91840a5a14258151a467","observation_id":"98a5edf3-3bec-4513-b1e0-5c1ecb7f32cc","resolution":{"observed_at":"2026-08-16T04:25:05.597081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.02981","last_updated":"2024-06-07T08:44:52Z","snapshot_observed_at":"2026-08-16T13:46:00.907207Z","submitted_at":"2024-06-05T06:23:49Z","title":"Local vs. Global Interpretability: A Computational Complexity Perspective","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.02981","snapshot_observed_at":"2026-08-16T04:25:05.601776Z","title":"Local vs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.601776Z"},"links":{"cited_paper":"/paper/2406.02981","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:702a19b05b91021c4e606cfd4fa37464c16323f87323a875d1f4b346279fa896","observation_id":"64e4beca-6ae9-4858-b525-195cbebf0549","resolution":{"observed_at":"2026-08-16T04:25:05.601776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1016/j.shpsc.2005.03.010","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.256151Z","title":"Explanation: A mechanist alternative","venue":null,"work_id":"2a6b7a91-6a28-4964-94fc-7bde02b791c4","year":2005},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.606747Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6277d537342469f37aa9486dace156911cb0f4aeeb56637fb8d7bd1110db27fe","observation_id":"3fa773a3-d922-4b64-81cd-e726fb950bf8","resolution":{"observed_at":"2026-08-16T04:25:06.262246Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.612261Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.612261Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:697d0bc31227800eaa832ec66e3af4226164b21caaa91e90877416417fdec797","observation_id":"3c2e7bee-d2fd-4024-bc7c-1cb2370aa993","resolution":{"observed_at":"2026-08-16T04:25:05.612261Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17805","last_updated":"2025-01-29T17:47:36Z","snapshot_observed_at":"2026-08-15T14:15:35.328292Z","submitted_at":"2025-01-29T17:47:36Z","title":"International AI Safety Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17805","snapshot_observed_at":"2026-08-16T04:25:05.617272Z","title":"International ai safety report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.617272Z"},"links":{"cited_paper":"/paper/2501.17805","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:fa7e7bf6982b7bc0dab17297fa0c8ac198c2aecd7caad97684e78ad2c29fd445","observation_id":"c90eb841-85a5-44f5-b15d-ee4d768cc0c8","resolution":{"observed_at":"2026-08-16T04:25:05.617272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14082","last_updated":"2024-08-23T23:02:28Z","snapshot_observed_at":"2026-07-06T18:03:38.397804Z","submitted_at":"2024-04-22T11:01:51Z","title":"Mechanistic Interpretability for AI Safety -- A Review","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14082","snapshot_observed_at":"2026-08-16T04:25:05.622447Z","title":"Mechanistic Interpretability for AI Safety -- A Review , April 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.622447Z"},"links":{"cited_paper":"/paper/2404.14082","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3d877d5dc5ef688facb967a6a9044c21f07a3a4d2d8e01356eaf6a63165dd191","observation_id":"ec496079-ff91-465c-b9db-71c5d491d4cc","resolution":{"observed_at":"2026-08-16T04:25:05.622447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2307/2180883","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.233516Z","title":null,"venue":null,"work_id":"23c0d7b2-0ec4-4d02-a0b0-039256e91932","year":1940},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.627697Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0ba42b3fa11de92126d85d42ae3378f22c0427630872e45e6fc7a38d3134d2f8","observation_id":"03f4349d-828d-465b-9065-1c382d45670d","resolution":{"observed_at":"2026-08-16T04:25:06.239659Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.632693Z","title":"Auditing local explanations is hard","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.632693Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c9785f4fef7f17ea1dee1c4d651859c0558463241efec72486140037f0d6a738","observation_id":"c960ebea-2022-4f48-a643-0b57db62aa68","resolution":{"observed_at":"2026-08-16T04:25:05.632693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.637441Z","title":"Language models can explain neurons in language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.637441Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3e8f43398454e88a618baa746aaf9991ba044309f4fa5191e49c839141dcdddd","observation_id":"ed2767c7-3100-4b87-beb9-87dce637ba3a","resolution":{"observed_at":"2026-08-16T04:25:05.637441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.07143","last_updated":"2021-04-14T22:04:48Z","snapshot_observed_at":"2026-08-17T09:17:27.672646Z","submitted_at":"2021-04-14T22:04:48Z","title":"An Interpretability Illusion for BERT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.07143","snapshot_observed_at":"2026-08-16T04:25:05.642287Z","title":"An Interpretability Illusion for BERT , April 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.642287Z"},"links":{"cited_paper":"/paper/2104.07143","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:8f2e64266f54c5ef050e61ba8ba954a269f6f964e7c6c261f1da9255e6b64556","observation_id":"87cc22af-9945-4683-b20e-ef73fe38ec6a","resolution":{"observed_at":"2026-08-16T04:25:05.642287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12241","last_updated":"2024-05-24T13:16:32Z","snapshot_observed_at":"2026-08-16T13:51:48.413485Z","submitted_at":"2024-05-17T17:03:46Z","title":"Identifying Functionally Important Features with End-to-End Sparse Dictionary Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12241","snapshot_observed_at":"2026-08-16T04:25:05.647126Z","title":"Identifying Functionally Important Features with End -to- End Sparse Dictionary Learning , May 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.647126Z"},"links":{"cited_paper":"/paper/2405.12241","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3623c6b0de0445fe4e9258f53ce282f068b279c3ed37ed51a6cfb0d0bf9b08fb","observation_id":"a2b41a39-4f83-4cfd-a9f0-3663c232c3da","resolution":{"observed_at":"2026-08-16T04:25:05.647126Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14926","last_updated":"2025-02-07T19:22:32Z","snapshot_observed_at":"2026-08-14T20:11:56.886595Z","submitted_at":"2025-01-24T21:31:12Z","title":"Interpretability in Parameter Space: Minimizing Mechanistic Description Length with Attribution-based Parameter Decomposition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14926","snapshot_observed_at":"2026-08-16T04:25:05.652281Z","title":"Interpretability in parameter space: Minimizing mechanistic description length with attribution-based parameter decomposition","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.652281Z"},"links":{"cited_paper":"/paper/2501.14926","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ee4cb30b8e9e32e2d16f6e3054e40c5a7a981a7e8d773a6f443020ee2517f944","observation_id":"3a76e277-6588-436b-b83d-5be24a5b09ec","resolution":{"observed_at":"2026-08-16T04:25:05.652281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.658022Z","title":"Towards Monosemanticity : Decomposing Language Models With Dictionary Learning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.658022Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:43e673a091501522dc87e80c3fcfacba9c55c8c2a7998277dfae027ddfc83273","observation_id":"135608c9-8774-44fb-bf3e-18692e656f90","resolution":{"observed_at":"2026-08-16T04:25:05.658022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.15740","last_updated":"2025-01-27T03:06:06Z","snapshot_observed_at":"2026-08-13T19:26:18.474749Z","submitted_at":"2025-01-27T03:06:06Z","title":"Propositional Interpretability in Artificial Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.15740","snapshot_observed_at":"2026-08-16T04:25:05.663766Z","title":"Propositional interpretability in artificial intelligence","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.663766Z"},"links":{"cited_paper":"/paper/2501.15740","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ff0001aacc60fbdd3117cd2389034adfc8e3d3948d6c37e2c861e42c378a5706","observation_id":"34747481-f21c-4452-9e82-01aac17a063e","resolution":{"observed_at":"2026-08-16T04:25:05.663766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.669143Z","title":"A toy model of universality: Reverse engineering how networks learn group operations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.669143Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3dd255a552ae7ac1440c5c417fdaec39ab0451cdcd8ca7e20d9bdc69c6724bf3","observation_id":"8c79eff4-85ec-4984-8618-b2188376c0c3","resolution":{"observed_at":"2026-08-16T04:25:05.669143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.674171Z","title":"The evolutionary origins of modularity","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.674171Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1e42f7b437c1b679e61a08f5819e4f22e235602ce15f0f32bd2200f5e45217c9","observation_id":"ab6e7c15-147c-49e3-a67e-2ee2331be54d","resolution":{"observed_at":"2026-08-16T04:25:05.674171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.14997","last_updated":"2023-10-28T20:05:52Z","snapshot_observed_at":"2026-08-16T15:36:54.698423Z","submitted_at":"2023-04-28T17:36:53Z","title":"Towards Automated Circuit Discovery for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.14997","snapshot_observed_at":"2026-08-16T04:25:05.678871Z","title":"Mavor-Parker, Aengus Lynch, Stefan Heimersheim, and Adrià Garriga-Alonso","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.678871Z"},"links":{"cited_paper":"/paper/2304.14997","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:e1e1bdee091b2042db00ee5c68e76ed0d633fa51cbf9e6803909c4be3985a912","observation_id":"0bbc4ffc-89aa-4a8a-9228-f21bfc00c894","resolution":{"observed_at":"2026-08-16T04:25:05.678871Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.684525Z","title":"Central dogma of molecular biology","venue":null,"work_id":null,"year":1970},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.684525Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:2fdb78aa8171dd685f1518091530aee2f1909c2ad58700387f316f88dd227ed4","observation_id":"288ced46-da29-40db-ae02-4244fd08479c","resolution":{"observed_at":"2026-08-16T04:25:05.684525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.690760Z","title":"The beginning of infinity: Explanations that transform the world","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.690760Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c42fa163313904aa8ff99871865c165ec2d031162720169518b26483253f8db6","observation_id":"0076b1cc-574b-4515-bbd6-8b68134604ec","resolution":{"observed_at":"2026-08-16T04:25:05.690760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.696683Z","title":null,"venue":null,"work_id":null,"year":1919},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.696683Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:160ddd74452d8fd9acda7155cda6ff8de1a17f06bc3312cf25cc4a040bb11efc","observation_id":"9e2e0718-96c4-4dee-82a6-24b4caa1f973","resolution":{"observed_at":"2026-08-16T04:25:05.696683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.702151Z","title":"Einstein","venue":null,"work_id":null,"year":1916},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.702151Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:02d2256dfcd87b822fb70f24e01b568220ff89463dc1306d4500320a04ba9186","observation_id":"1643f022-9fd7-4021-af3f-81f37f4af433","resolution":{"observed_at":"2026-08-16T04:25:05.702151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.707415Z","title":null,"venue":null,"work_id":null,"year":1974},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.707415Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3ce3adcbbf269417c560424d02d23f026d30fb96fb514b557d2d27931ef3bd83","observation_id":"6badf7f6-0bd7-48dd-b6d1-dbc160e7f37b","resolution":{"observed_at":"2026-08-16T04:25:05.707415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.03386","last_updated":"2021-03-04T23:53:53Z","snapshot_observed_at":"2026-08-16T18:41:07.307930Z","submitted_at":"2021-03-04T23:53:53Z","title":"Clusterability in Neural Networks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.03386","snapshot_observed_at":"2026-08-16T04:25:05.712412Z","title":"Clusterability in neural networks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.712412Z"},"links":{"cited_paper":"/paper/2103.03386","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:85d5f0b94db8bb3389b8ab96a4d41718639c42659e88b9038a3feee84c6bf5ed","observation_id":"465f48a9-16d1-4854-9556-2b1edce37af5","resolution":{"observed_at":"2026-08-16T04:25:05.712412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.718000Z","title":"Interpretability illusions in the generalization of simplified models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.718000Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:51a021447c2ad4c3ec9e2a2071aafc4c723c6bc0cea2df0b396fc33998adba5a","observation_id":"d72172f6-6ee7-4cf2-a87e-c8aa458a6a2e","resolution":{"observed_at":"2026-08-16T04:25:05.718000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04093","last_updated":"2024-06-06T14:10:12Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-06T14:10:12Z","title":"Scaling and evaluating sparse autoencoders","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04093","snapshot_observed_at":"2026-08-16T04:25:05.723010Z","title":"Scaling and evaluating sparse autoencoders, June 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.723010Z"},"links":{"cited_paper":"/paper/2406.04093","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:7480711385949b1eb0d9262e4a62176ffc6afccced0e1b15156d6da03e959f3a","observation_id":"f97a349d-416e-4ec6-ba95-63ed4aa9bd7f","resolution":{"observed_at":"2026-08-16T04:25:05.723010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.728174Z","title":"Causal abstractions of neural networks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.728174Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:110e0a46cd0af0b004780878a816deaadd39d1e24e5e426f6cc83e0f06ee7169","observation_id":"712695dc-797c-4577-a954-6c0f803ec29b","resolution":{"observed_at":"2026-08-16T04:25:05.728174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.04709","last_updated":"2025-05-08T23:00:37Z","snapshot_observed_at":"2026-08-16T16:03:00.653173Z","submitted_at":"2023-01-11T20:42:41Z","title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.04709","snapshot_observed_at":"2026-08-16T04:25:05.733991Z","title":"Causal abstraction for faithful model interpretation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.733991Z"},"links":{"cited_paper":"/paper/2301.04709","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6b4aa2f0b28357825fd0c0953a4abff9d0e1e2efd8566caead233fc74769d5a8","observation_id":"22aea982-0987-44f7-ab5b-592ababe55dd","resolution":{"observed_at":"2026-08-16T04:25:05.733991Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.740650Z","title":"Clustering algorithms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.740650Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1e06fc6961b06736071110a383525202ad00d49b8b79781ff288f9ea33388cf7","observation_id":"25eafe53-c2b4-44c6-a494-fd448e1bf91c","resolution":{"observed_at":"2026-08-16T04:25:05.740650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.746673Z","title":"Compact proofs of model performance via mechanistic interpretability","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.746673Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6ba77aaac378e7e585f231fc663ef2f085e97da612d75da3c66e952e05521a51","observation_id":"5d25dd86-a4ae-4bdc-bad7-135601377bcb","resolution":{"observed_at":"2026-08-16T04:25:05.746673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.870126Z","title":"Interpbench: Semi-synthetic transformers for evaluating mechanistic interpretability techniques","venue":null,"work_id":"e54e80b9-a160-4d8b-b2a4-7bf9d719c953","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.752590Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ccf54bd5b09726532ceab279f8e746bbfe008774088c883a55e16f9ec372d64b","observation_id":"33317061-ea4a-44d4-a1cc-f3220fce6bba","resolution":{"observed_at":"2026-08-16T04:25:07.875795Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.851429Z","title":"Hastie, R","venue":null,"work_id":"10fa1ef4-4cd8-4c81-9b61-59137ecab1d4","year":2009},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.757293Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:d97b8f06b5c553de80c007631a19a83b26d48563391eddf6379720fb10a378e5","observation_id":"6c1fbada-0a31-460d-8864-8ab9e751bc15","resolution":{"observed_at":"2026-08-16T04:25:07.856843Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.830083Z","title":"Hempel and Paul Oppenheim","venue":null,"work_id":"f52e96cd-5c86-4140-aca9-b1954c633af1","year":1948},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.762261Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:9795f24dd4cca22deef6f11996877def9cea37f34ef73e72b95cec1e31d28f6d","observation_id":"6bb4f5b1-e9b0-4de3-b18b-8a9922be0e2e","resolution":{"observed_at":"2026-08-16T04:25:07.837001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.811267Z","title":"Philosophy of Natural Science","venue":null,"work_id":"95ab8c89-9a9e-4ddd-877c-0eb1ce95b61a","year":1966},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.767579Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:86d012bfaf87b9d65aa7e82feb379b17744002b3d9642baf0b58589fc24ad78a","observation_id":"0fc0361d-3d60-41e5-9e5c-c8186b9233c4","resolution":{"observed_at":"2026-08-16T04:25:07.816805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.791761Z","title":"Bayesianism and inference to the best explanation","venue":null,"work_id":"99363bfb-e214-486a-925d-3bdc68758351","year":2014},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.773005Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a48c477fb181a264d7ec0b787627a59ccc90737291e25b3e8b43021d4e2e38fd","observation_id":"e355e61b-deec-4081-8248-fe70d5984b8b","resolution":{"observed_at":"2026-08-16T04:25:07.797867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02364","last_updated":"2025-08-01T09:20:40Z","snapshot_observed_at":"2026-08-16T14:21:46.708146Z","submitted_at":"2024-02-04T06:23:05Z","title":"Loss Landscape Degeneracy and Stagewise Development in Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02364","snapshot_observed_at":"2026-08-16T04:25:05.779468Z","title":"The developmental landscape of in-context learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.779468Z"},"links":{"cited_paper":"/paper/2402.02364","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3f56c55f2d3e3fc6ea256ed5e451c266d44723cc82973ba4582c401b51dc167c","observation_id":"b0510d13-33cf-4cb9-bfb5-b1bc7a98b276","resolution":{"observed_at":"2026-08-16T04:25:05.779468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.785160Z","title":"Sparse autoencoders find highly interpretable features in language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.785160Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f6b2e766252f448ce6f40048ed6197b8e51643523347a4684059080e0193e007","observation_id":"c5ec30f4-92f1-40f8-9acf-a47b1637be70","resolution":{"observed_at":"2026-08-16T04:25:05.785160Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.762063Z","title":"Hutter, E","venue":null,"work_id":"e762c19b-f9c8-4491-9749-ee475bc854f3","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.791367Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:99cdf52a58d8ef0e4421fa67874889d434e020c4207cdb7e0d6e31cdcc7981b5","observation_id":"40da69cb-7913-46ec-9031-94e2f0f81322","resolution":{"observed_at":"2026-08-16T04:25:07.767475Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.743207Z","title":"Fine-tuning neural networks to match their interpretation: Towards scaling compact proofs, 2025","venue":null,"work_id":"8877e86b-c974-4766-a0a8-7b338ad07725","year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.796480Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:4432f00afd746131f76661900a009f7a59fe77bac2cdd5f4fbffe051682a7096","observation_id":"1c3adc74-c6de-4fdf-b320-a17fec806d6d","resolution":{"observed_at":"2026-08-16T04:25:07.749269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.717970Z","title":"Kandel, J.H","venue":null,"work_id":"bb27f722-1eb1-4bf5-a0d2-7417f61e3b44","year":2000},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.801635Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:81cb02fe4018cf809c8fd2517f5820892d0646ace5355564b43bb74c6552404e","observation_id":"7ff9effe-ae4b-46a8-be2e-3ce3887cac71","resolution":{"observed_at":"2026-08-16T04:25:07.729487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.699617Z","title":"Saebench: a comprehensive benchmark for sparse autoencoders, 2024","venue":null,"work_id":"5a4c3359-9828-474a-9cf3-805cb797679e","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.807554Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:5df0623d2d211b0815332de155ca6b8871ae2d32dd989fbe76075f4b9337fa08","observation_id":"6d585e48-eb4a-4bc0-9aac-d340d76ab5f0","resolution":{"observed_at":"2026-08-16T04:25:07.705111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.681692Z","title":"Kennefick","venue":null,"work_id":"da9e08bd-579b-4c35-b0a4-c3081af48560","year":1919},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.814962Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f6fadee37e0cee49ff6d9a57ddb521a2779ef2e3c790291d680176a4edee432d","observation_id":"fd84da5c-f9e0-43b8-96b3-ed972dd73d61","resolution":{"observed_at":"2026-08-16T04:25:07.687071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.661190Z","title":"Explanatory unification","venue":null,"work_id":"9b1f8f18-7195-4bc6-9168-7fd53a46d205","year":1981},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.820191Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b824fbaa74e42406fc8c7f46726a677814392bd6f3b2c51be313184ca6cf16ec","observation_id":"16b197c7-a52f-41ad-bccc-66f48285f750","resolution":{"observed_at":"2026-08-16T04:25:07.667802Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.639413Z","title":"Three approaches to the quantitative definition ofinformation","venue":null,"work_id":"3cd4f4b2-2ab8-4786-b625-46fd32db056c","year":1965},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.825015Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:82dcb91433e68819220fb5a04b0151a80e0a1db392261c8c8c02cc2843f743fa","observation_id":"10e71510-9e4a-4a0e-ad3c-0aaf7a4baba4","resolution":{"observed_at":"2026-08-16T04:25:07.646435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.622120Z","title":null,"venue":null,"work_id":"71934480-c0fb-49d2-9cd5-378330a1027e","year":1981},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.830341Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:55dc15fb1d9a6ab07f1af6ca279c3ce88f2d5725733b9db7c8752542d46f838c","observation_id":"17746ef1-137a-4c4f-b7fd-8d3fa293643b","resolution":{"observed_at":"2026-08-16T04:25:07.627680Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.598916Z","title":"The Structure of Scientific Revolutions","venue":null,"work_id":"14dd557c-c095-45d3-9e1e-d6135eed97c9","year":1962},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.835464Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ef83e45cd3ec4da31b079808361c871e1645f9bb5e47b38563774a945325ca15","observation_id":"1587381a-db0b-4c29-8c54-0a0c152ca444","resolution":{"observed_at":"2026-08-16T04:25:07.607038Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.580037Z","title":"Falsification and the methodology of scientific research programmes","venue":null,"work_id":"2331be14-3ed4-4689-a41c-f0a22bfb8b75","year":1970},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.840481Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3e913b0864f1cda0fe37acdb8d5c52ce9fc19debb3da80a0ced6cdc6a4b1aebd","observation_id":"aaa33f02-0ff5-4a53-aada-a5d9be8f6d4f","resolution":{"observed_at":"2026-08-16T04:25:07.586373Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.562069Z","title":"The Methodology of Scientific Research Programmes","venue":null,"work_id":"13d98109-e392-4f1a-b37b-fd049ab94603","year":1978},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.845034Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a09a00f6dee2521eb2cfcfc0b8c11b7483ed9d01eb8e5a7a2861338035597af6","observation_id":"91c964fc-0773-420c-8279-fc1eb53403b3","resolution":{"observed_at":"2026-08-16T04:25:07.567371Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.04878","last_updated":"2025-02-07T12:33:08Z","snapshot_observed_at":"2026-08-13T12:18:38.447222Z","submitted_at":"2025-02-07T12:33:08Z","title":"Sparse Autoencoders Do Not Find Canonical Units of Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.04878","snapshot_observed_at":"2026-08-16T04:25:05.850864Z","title":"Sparse autoencoders do not find canonical units of analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.850864Z"},"links":{"cited_paper":"/paper/2502.04878","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:5135a22116334b3f6b8f2d2a75155e7746cae76497bf744b2783b3bef3550511","observation_id":"f9885b31-5dd2-4dae-97f1-f880fb428f8b","resolution":{"observed_at":"2026-08-16T04:25:05.850864Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.856242Z","title":"Lindsay and David Bau","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.856242Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:eaf4883cbc1bdcf37ff00d55603e30b1b2a2d7cde215b29f1f9da4439cb7faf8","observation_id":"f7a9f288-c896-4fba-9c44-d8753e20975f","resolution":{"observed_at":"2026-08-16T04:25:05.856242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.862843Z","title":"The mythos of model interpretability: In machine learning, the concept of interpretability is both important and slippery","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.862843Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c475e0871d1ad448dc977eb39b38d264fefbf66f312f7858a16afbb9452be641","observation_id":"b0ae1407-4a60-447c-8d2e-dada1f82ccfc","resolution":{"observed_at":"2026-08-16T04:25:05.862843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.533799Z","title":"Mechanistic mode connectivity","venue":null,"work_id":"26377831-d4e0-4b0d-b5cd-708f55fdbaac","year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.868951Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ca0f847454c7ae4a9d83c4dde7edff53f1a6c0e294133da89a1bb1db8efe4090","observation_id":"9eb3975f-8e1f-4e66-a474-ec9c3b1a62af","resolution":{"observed_at":"2026-08-16T04:25:07.539180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.874084Z","title":"Information theory, inference and learning algorithms","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.874084Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c2ec497ad98e15cd18adf296cc79ec1bb450715ee9257cf61d556e87865906ca","observation_id":"ca10ed64-93ed-4ac4-a197-12840824058c","resolution":{"observed_at":"2026-08-16T04:25:05.874084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.501630Z","title":"Is this the subspace you are looking for? an interpretability illusion for subspace activation patching","venue":null,"work_id":"e7ed6346-bcf8-4546-8016-152a5e6a7e55","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.879925Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:66a36af7c84d3b11b0cbf9ca5c44b7d4b9becf5f2200cf1c44c0250e4c056f73","observation_id":"86793f78-8c9c-49a5-b6b4-0c5e352fa2df","resolution":{"observed_at":"2026-08-16T04:25:07.507881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.481672Z","title":"Downstream applications as validation of interpretability progress, March 2025","venue":null,"work_id":"7cc98727-fb6a-46cf-8d2c-f0eba287fccf","year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.885763Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:79481f1a4bdb3933d273db7a301050a18f3b066cdea27059ea2d634cc2137185","observation_id":"2085c2d2-ad9b-49d1-9c1b-412de83b2a98","resolution":{"observed_at":"2026-08-16T04:25:07.486729Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19647","last_updated":"2025-03-27T05:44:45Z","snapshot_observed_at":"2026-08-14T07:35:01.333549Z","submitted_at":"2024-03-28T17:56:07Z","title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19647","snapshot_observed_at":"2026-08-16T04:25:05.891673Z","title":"Michaud, Yonatan Belinkov, David Bau, and Aaron Mueller","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.891673Z"},"links":{"cited_paper":"/paper/2403.19647","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:5da2bba86349e1bdfa4ed0e32944d60a74dbc52e597822b152a444c5d078b7f9","observation_id":"484ea65f-aec7-463d-80ea-4d5c9dc4fd29","resolution":{"observed_at":"2026-08-16T04:25:05.891673Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.463184Z","title":"Cognitive styles in two cognitive sciences","venue":null,"work_id":"a95f36a6-b00e-48b4-a249-8ba993ee1299","year":2012},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.898471Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1615bf0f1c597dc87e23a483fad2724fad9e4017e65e778d19a7845faa552805","observation_id":"eb401795-6468-4d96-8e49-cfb9a832798b","resolution":{"observed_at":"2026-08-16T04:25:07.468750Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.905678Z","title":"Zoom in: An introduction to circuits","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.905678Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:0bfc9a7bd37e96fe55cbbf17437c14bd16daa4c560cd15e1424aaf58a2f0240a","observation_id":"fb3e9be1-3e90-4402-bf7a-be0ded544915","resolution":{"observed_at":"2026-08-16T04:25:05.905678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.11895","last_updated":"2022-09-24T00:43:19Z","snapshot_observed_at":"2026-08-15T09:43:59.961298Z","submitted_at":"2022-09-24T00:43:19Z","title":"In-context Learning and Induction Heads","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.11895","snapshot_observed_at":"2026-08-16T04:25:05.911761Z","title":"In-context learning and induction heads","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.911761Z"},"links":{"cited_paper":"/paper/2209.11895","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6cb3ff15d8ab7c1cce6639b566ea5822f12d9fa332409f391bb225a9e5d5eb14","observation_id":"536f7abc-d582-4db8-9057-fae24a5dba45","resolution":{"observed_at":"2026-08-16T04:25:05.911761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.13928","last_updated":"2025-08-06T13:47:10Z","snapshot_observed_at":"2026-08-16T13:08:17.757544Z","submitted_at":"2024-10-17T17:56:01Z","title":"Automatically Interpreting Millions of Features in Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.13928","snapshot_observed_at":"2026-08-16T04:25:05.918441Z","title":"Automatically interpreting millions of features in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.918441Z"},"links":{"cited_paper":"/paper/2410.13928","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:6d38d3134899e22ad66b44191059f359a98fedeb75264d6d16d85ff74211ecdd","observation_id":"293d1e50-1950-4e08-84f3-51ad6d5bb269","resolution":{"observed_at":"2026-08-16T04:25:05.918441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.925700Z","title":"Causality","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.925700Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:e1994f9eebb85368290cd2958e863fffe9030123f8a8eebc396b4f713562b2ee","observation_id":"3b0b26c7-9cc4-4b96-8dd8-8f02df58426d","resolution":{"observed_at":"2026-08-16T04:25:05.925700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.410379Z","title":"Poincar \\'e","venue":null,"work_id":"caa93d46-3962-447f-afa1-4017648bbfaf","year":1905},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.932748Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:f0f90e6cadbb8106d7a1f4fa20c4d518b399af95545bd4f55ddf26cfe6095c88","observation_id":"c6aebdff-6c3d-404e-b0fd-1062e6f53578","resolution":{"observed_at":"2026-08-16T04:25:07.416845Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.944080Z","title":null,"venue":null,"work_id":null,"year":1935},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.944080Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:b0d1b7f8d0e0e6e5eb67567fc9aef07879ac0e3af29dde1b303c5afe52bbad13","observation_id":"e2af5c31-1409-4f9f-b0c2-e63b82fc5688","resolution":{"observed_at":"2026-08-16T04:25:05.944080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.3998/phimp.1521","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.216159Z","title":"Hume on theoretical simplicity","venue":null,"work_id":"da413d8b-0960-4631-953f-4133a331a693","year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.951779Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:543b474ef81216703adba1c0c6e0eef61645144016e9738d8c905c32db614ef1","observation_id":"5b60c837-b55c-4963-8912-cee876b825cf","resolution":{"observed_at":"2026-08-16T04:25:06.221515Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.962513Z","title":"Escalation risks from language models in military and diplomatic decision-making","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.962513Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:a10d5e1296502164b9dfe1370310ff48b44495d3c6b5785d3266eb4d6fdcb365","observation_id":"e7de4f8c-c5f8-4c6e-bdb3-6d566e58bd11","resolution":{"observed_at":"2026-08-16T04:25:05.962513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.378688Z","title":"Four decades of scientific explanation","venue":null,"work_id":"fb44d7a8-d8f2-49fb-98a2-f178187b2846","year":1989},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.968237Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:ff2f3a2decb94df85e2fb940533d903becb87e43b5487ea78fc039f1e62641eb","observation_id":"4747c415-0ae3-4b04-b1ee-2029e9bd4a1f","resolution":{"observed_at":"2026-08-16T04:25:07.384390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.358969Z","title":null,"venue":null,"work_id":"9665ca3d-638b-4587-aede-53acc4fff13f","year":1984},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.973050Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:3477f899565d655b4a6897b43d305b4bbf1d9d1f37b47874e485dfce8223d556","observation_id":"80943169-dfbc-466b-a923-efe22d9afad1","resolution":{"observed_at":"2026-08-16T04:25:07.365849Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09087","last_updated":"2024-10-07T15:02:12Z","snapshot_observed_at":"2026-08-16T13:11:52.807313Z","submitted_at":"2024-10-07T15:02:12Z","title":"Mechanistic?","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.09087","snapshot_observed_at":"2026-08-16T04:25:05.978424Z","title":"Mechanistic?, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.978424Z"},"links":{"cited_paper":"/paper/2410.09087","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:80bd5e983494ceb0291c24fa8ff971d2a7f54d0e20ea08bfff9a3d2210bbfa1d","observation_id":"29c5b0f6-51c4-4735-99a3-2c9f05b2f048","resolution":{"observed_at":"2026-08-16T04:25:05.978424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.339909Z","title":"Theoretical Virtues in Science: Uncovering Reality Through Theory","venue":null,"work_id":"6f56e31d-b057-47a8-a0b0-781e12bf3633","year":2018},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.984022Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:29212002a0ef24f5e590aab5fd3310b8e9f91d0bec7735beb850b2aac179f54f","observation_id":"08075a01-c91b-4cc8-8e58-12d0004ac73c","resolution":{"observed_at":"2026-08-16T04:25:07.346087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.990103Z","title":"Riechers, Lucas Teixeira, Alexander Gietelink Oldenziel, and Sarah Marzen","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.990103Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:dbb2fc959a7f0e6c28e1b4bb8363c0770ae0341bcf528d8f31a3a361a6b1b07a","observation_id":"740ce7be-8b7a-4b84-b752-dd27ef2f759c","resolution":{"observed_at":"2026-08-16T04:25:05.990103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:05.996982Z","title":"A mathematical theory of communication","venue":null,"work_id":null,"year":1948},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:05.996982Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:35b148139737771143033dd18fec3bc191f9411dbe76f9db8d0654994698fb12","observation_id":"1b7789d5-9c8b-4ace-9512-ac09ab787734","resolution":{"observed_at":"2026-08-16T04:25:05.996982Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.16496","last_updated":"2025-01-27T20:57:18Z","snapshot_observed_at":"2026-07-06T20:27:04.664873Z","submitted_at":"2025-01-27T20:57:18Z","title":"Open Problems in Mechanistic Interpretability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.16496","snapshot_observed_at":"2026-08-16T04:25:06.003123Z","title":"Michaud, Stephen Casper, Max Tegmark, William Saunders, David Bau, Eric Todd, Atticus Geiger, Mor Geva, Jesse Hoogland, Daniel Murfet, and Tom McGrath","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.003123Z"},"links":{"cited_paper":"/paper/2501.16496","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:785a805c0f22e09d076a699963f32da7c394ccbae367c4d55cb1c4fd8bd0de09","observation_id":"b417d3b7-a3fc-4a92-a138-97465a1fe086","resolution":{"observed_at":"2026-08-16T04:25:06.003123Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.290004Z","title":"Hypothesis testing the circuit hypothesis in LLM s","venue":null,"work_id":"ec2f18a9-e756-41d2-8ad4-0a2ae0a5bd50","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.010034Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:e80e21850ca1356b43f3e232083266fad75942c2dd03ee5df5436eecb2fbf0a8","observation_id":"e07c0ebc-f326-427a-a03f-bbc9cbaf8612","resolution":{"observed_at":"2026-08-16T04:25:07.296737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.272093Z","title":"The golden mean of scientific virtues, 2024","venue":null,"work_id":"ee1a0dc9-ed96-40ab-9eda-eb57113de866","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.016375Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:2bbe7d9bef019a7ee427134e28deb61411a79deedaccc57646a277418e466b98","observation_id":"1d512897-3ac1-4f50-b804-40995ead7c76","resolution":{"observed_at":"2026-08-16T04:25:07.276920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.253076Z","title":"Knowledge in Perspective: Selected Essays in Epistemology","venue":null,"work_id":"5548e441-35e6-4b62-a6c5-029f65b7448c","year":1991},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.024761Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:cc1ed38cb15810bf50191de1dfc8b31dc3f8aa9ab692725a5c9160441d765020","observation_id":"1d7ce50c-4f21-4f8c-a2d0-82bfdc5f1ff6","resolution":{"observed_at":"2026-08-16T04:25:07.259403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.233793Z","title":"Grokking group multiplication with cosets","venue":null,"work_id":"2fe8047a-1207-4fd2-b883-1e4d7b92568c","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.031559Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:4495bfad6430709f7871790fb4a0b424f34d939c7507ac0c6df078cdaa6b4a63","observation_id":"352006f4-c9b9-4bf7-bc3c-71372f66f6b8","resolution":{"observed_at":"2026-08-16T04:25:07.240232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.212737Z","title":"Simplicity as Evidence of Truth","venue":null,"work_id":"65c882b4-dcf0-4f1b-919c-b43e6aa19d5a","year":1997},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.037028Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c9643b22a1cac96c680ad58b7575fd70f4870d32e195344c5b49481e0b8f899e","observation_id":"ca0eab37-098d-4e05-a20d-024b68c63bb8","resolution":{"observed_at":"2026-08-16T04:25:07.218842Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.13714","last_updated":"2024-09-07T10:02:51Z","snapshot_observed_at":"2026-08-16T13:20:33.519678Z","submitted_at":"2024-09-07T10:02:51Z","title":"TracrBench: Generating Interpretability Testbeds with Large Language Models","version":1},"cited_work":{"arxiv_id":"2409.13714","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.13714","snapshot_observed_at":"2026-08-16T04:25:06.323657Z","title":"TracrBench: Generating Interpretability Testbeds with Large Language Models","venue":"cs.CL","work_id":"24f26777-fae9-4acf-a74a-3367b6104d79","year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.043348Z"},"links":{"cited_paper":"/paper/2409.13714","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:bc35d5c848623666ae182b2195291adfbd0a02113cc4967e965dcb23aeab7c41","observation_id":"2e2cb470-0087-4b83-a9c2-b26796256c1f","resolution":{"observed_at":"2026-08-16T04:25:06.330001Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.185289Z","title":"Interpretability in the wild: a circuit for indirect object identification in gpt-2 small","venue":null,"work_id":"01114bac-7a99-4508-abf8-22ec661a5a7a","year":2023},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.049199Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:4c66c37d34ad7bfc16a080e8a06a7621e97873b87182276bca5de360b871db79","observation_id":"0a2c7ecf-696a-4a1a-9927-dc96616081c7","resolution":{"observed_at":"2026-08-16T04:25:07.193035Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1093/analys/65.3.205","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.194033Z","title":"Why favour simplicity? Analysis, 65 0 (3): 0 205--210, 2005","venue":null,"work_id":"7330de9a-0f1b-49d5-a62c-3a6f6cdbd97e","year":2005},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.057524Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:bb9d70d6e14e1bf335c58909ba2696023741c761c9ed0cb69f2327c18d750d9a","observation_id":"57fadde9-0811-439f-84ad-fa1f01846649","resolution":{"observed_at":"2026-08-16T04:25:06.202634Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.167522Z","title":"Understanding as compression","venue":null,"work_id":"74dbc341-e9f4-4760-833f-240ee97d2ba4","year":2019},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.064714Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:8b0df09df25d566537e85ed81ef4f2df21c5507fab0506f16d34ff5b79400c3c","observation_id":"b37e3d23-7ad3-4fdd-b774-1a560c33b6d1","resolution":{"observed_at":"2026-08-16T04:25:07.173192Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.148654Z","title":"Geschichte und Naturwissenschaft","venue":null,"work_id":"c70f713d-4f40-4cc9-ac2f-e023b2b492aa","year":null},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.071370Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:65bb29a33ea8d5a49292b3b7bc69d5b0add290881d2bbe1cf40c803cb85d58c1","observation_id":"095fdfb3-67ab-48ae-a712-b6d6f312a35d","resolution":{"observed_at":"2026-08-16T04:25:07.154724Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.129771Z","title":"From probability to consilience: How explanatory values implement bayesian reasoning","venue":null,"work_id":"e58874bb-1f35-454d-8f79-af96acc32162","year":2020},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.076238Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1f29094e30e1d992b2eedfea3340bffc97681326629e1d336733c1eed8beab53","observation_id":"721adab6-90b7-48eb-81f9-c03e9c9c1263","resolution":{"observed_at":"2026-08-16T04:25:07.136110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.105582Z","title":"Woodward","venue":null,"work_id":"03d48edb-cadf-4807-a581-dda5eacc8729","year":2003},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.081518Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:98c103bf4d22dec259b9a514b50d2764291590ce4f76932289549ffbf00b3b3b","observation_id":"0c046019-3c7f-41e4-9991-f9a048c1a470","resolution":{"observed_at":"2026-08-16T04:25:07.111381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07476","last_updated":"2025-01-24T23:41:37Z","snapshot_observed_at":"2026-08-16T13:10:54.838319Z","submitted_at":"2024-10-09T23:02:00Z","title":"Towards a unified and verified understanding of group-operation networks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07476","snapshot_observed_at":"2026-08-16T04:25:06.086534Z","title":"Unifying and verifying mechanistic interpretations: A case study with group operations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.086534Z"},"links":{"cited_paper":"/paper/2410.07476","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:9dac5de6bc7d35edc2a5f044a4f90e7a6104df8d9e4f710bb203634dc3ecb465","observation_id":"3ff95413-4b01-4574-8af5-5fffe4e5c756","resolution":{"observed_at":"2026-08-16T04:25:06.086534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17148","last_updated":"2025-03-03T21:15:30Z","snapshot_observed_at":"2026-08-16T12:58:32.947185Z","submitted_at":"2025-01-28T18:51:24Z","title":"AxBench: Steering LLMs? Even Simple Baselines Outperform Sparse Autoencoders","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.17148","snapshot_observed_at":"2026-08-16T04:25:06.093358Z","title":"Manning, and Christopher Potts","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.093358Z"},"links":{"cited_paper":"/paper/2501.17148","citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:1461fb4e39ed7d37d47550cf42037718b19eddef82fe36bcf63c7c40c0215ecb","observation_id":"74a24061-0e17-4599-a309-ec652df7de8d","resolution":{"observed_at":"2026-08-16T04:25:06.093358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:06.101369Z","title":"A theory of usable information under computational constraints","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.101369Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:c0164485e13feddc806fc4463d5c484b45dbd20ac2ac85851537a9545400815e","observation_id":"5268333d-c642-4bbb-b2eb-f957abac149a","resolution":{"observed_at":"2026-08-16T04:25:06.101369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T04:25:07.070612Z","title":"Locally decodable codes","venue":null,"work_id":"44c36a21-d59d-4b9d-b0ca-7336a3b57e73","year":2012},"citing_paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","version":1},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-16T04:25:06.108473Z"},"links":{"citing_paper":"/paper/2505.01372"},"observation_digest":"sha256:8a42b30a7738fdfe91bb91abaff38d8736feba52e1438309eb929ca7a0691d34","observation_id":"7474a404-cacd-4e6c-bd4e-726f49f34e19","resolution":{"observed_at":"2026-08-16T04:25:07.076567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.01372","last_updated":"2025-05-02T16:18:40Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T10:12:48.459153Z","submitted_at":"2025-05-02T16:18:40Z","title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":62,"verified_exact":5,"verified_fuzzy":33},"total_outbound_references":105},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 100 of 105 outbound references and 1 inbound Pith citation observation for arXiv:2505.01372."}