{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OD2VQQRIZWS2MYA74IZ7W3BI52","short_pith_number":"pith:OD2VQQRI","schema_version":"1.0","canonical_sha256":"70f5584228cda5a6601fe233fb6c28eebf61c6ac0600f9fba739a5a3b98e0b62","source":{"kind":"arxiv","id":"2409.13714","version":1},"attestation_state":"computed","paper":{"title":"TracrBench: Generating Interpretability Testbeds with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hannes Thurnherr, J\\'er\\'emy Scheurer","submitted_at":"2024-09-07T10:02:51Z","abstract_excerpt":"Achieving a mechanistic understanding of transformer-based language models is an open challenge, especially due to their large number of parameters. Moreover, the lack of ground truth mappings between model weights and their functional roles hinders the effective evaluation of interpretability methods, impeding overall progress. Tracr, a method for generating compiled transformers with inherent ground truth mappings in RASP, has been proposed to address this issue. However, manually creating a large number of models needed for verifying interpretability methods is labour-intensive and time-con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13714","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-07T10:02:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6824c497fd98ec67119b580b4e80533d88ee2adf4698406731792d334fd6f7cf","abstract_canon_sha256":"0cb4d2e49cadf60d6d31f57fb7d92ad6553d54cecaeb7abde5629b6eb5180958"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:10:05.887452Z","signature_b64":"XYBQR7Jb1udOk1WHhqnUt+Yj3Bp38bnDHRuxsTLmmwRxy0LY1MbgZ4bcAiSoz1boxbaZBrwYleCCjgsg7hl5AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"70f5584228cda5a6601fe233fb6c28eebf61c6ac0600f9fba739a5a3b98e0b62","last_reissued_at":"2026-07-05T09:10:05.886970Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:10:05.886970Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TracrBench: Generating Interpretability Testbeds with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hannes Thurnherr, J\\'er\\'emy Scheurer","submitted_at":"2024-09-07T10:02:51Z","abstract_excerpt":"Achieving a mechanistic understanding of transformer-based language models is an open challenge, especially due to their large number of parameters. Moreover, the lack of ground truth mappings between model weights and their functional roles hinders the effective evaluation of interpretability methods, impeding overall progress. Tracr, a method for generating compiled transformers with inherent ground truth mappings in RASP, has been proposed to address this issue. However, manually creating a large number of models needed for verifying interpretability methods is labour-intensive and time-con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13714","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13714/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13714","created_at":"2026-07-05T09:10:05.887027+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13714v1","created_at":"2026-07-05T09:10:05.887027+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13714","created_at":"2026-07-05T09:10:05.887027+00:00"},{"alias_kind":"pith_short_12","alias_value":"OD2VQQRIZWS2","created_at":"2026-07-05T09:10:05.887027+00:00"},{"alias_kind":"pith_short_16","alias_value":"OD2VQQRIZWS2MYA7","created_at":"2026-07-05T09:10:05.887027+00:00"},{"alias_kind":"pith_short_8","alias_value":"OD2VQQRI","created_at":"2026-07-05T09:10:05.887027+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.01372","citing_title":"Evaluating Explanations: An Explanatory Virtues Framework for Mechanistic Interpretability -- The Strange Science Part I.ii","ref_index":90,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52","json":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52.json","graph_json":"https://pith.science/api/pith-number/OD2VQQRIZWS2MYA74IZ7W3BI52/graph.json","events_json":"https://pith.science/api/pith-number/OD2VQQRIZWS2MYA74IZ7W3BI52/events.json","paper":"https://pith.science/paper/OD2VQQRI"},"agent_actions":{"view_html":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52","download_json":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52.json","view_paper":"https://pith.science/paper/OD2VQQRI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13714&json=true","fetch_graph":"https://pith.science/api/pith-number/OD2VQQRIZWS2MYA74IZ7W3BI52/graph.json","fetch_events":"https://pith.science/api/pith-number/OD2VQQRIZWS2MYA74IZ7W3BI52/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52/action/storage_attestation","attest_author":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52/action/author_attestation","sign_citation":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52/action/citation_signature","submit_replication":"https://pith.science/pith/OD2VQQRIZWS2MYA74IZ7W3BI52/action/replication_record"}},"created_at":"2026-07-05T09:10:05.887027+00:00","updated_at":"2026-07-05T09:10:05.887027+00:00"}