{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:RH2WQF6QVY3IOSFI3TZUKOBXIS","short_pith_number":"pith:RH2WQF6Q","canonical_record":{"source":{"id":"2506.02204","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:44:06Z","cross_cats_sorted":[],"title_canon_sha256":"a99b2b8f0417ceb5c99a559ab71a9d90b90aac5a3ea9325376f915df128d7069","abstract_canon_sha256":"04f60d0b2643e4d5ab3c3d532637037693ce87a99ba495afc79e803957e35114"},"schema_version":"1.0"},"canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","source":{"kind":"arxiv","id":"2506.02204","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.02204","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"arxiv_version","alias_value":"2506.02204v2","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02204","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"pith_short_12","alias_value":"RH2WQF6QVY3I","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"pith_short_16","alias_value":"RH2WQF6QVY3IOSFI","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"pith_short_8","alias_value":"RH2WQF6Q","created_at":"2026-07-05T11:18:43Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:RH2WQF6QVY3IOSFI3TZUKOBXIS","target":"record","payload":{"canonical_record":{"source":{"id":"2506.02204","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:44:06Z","cross_cats_sorted":[],"title_canon_sha256":"a99b2b8f0417ceb5c99a559ab71a9d90b90aac5a3ea9325376f915df128d7069","abstract_canon_sha256":"04f60d0b2643e4d5ab3c3d532637037693ce87a99ba495afc79e803957e35114"},"schema_version":"1.0"},"canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:43.401074Z","signature_b64":"2uLgnXXtQXzKkvaQuCw1EDbphKBASKMMvaHW4QTwatVugB7w9reBcuwUmfQDGm0DZdW/aRK3zYhKlIASFYI4DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","last_reissued_at":"2026-07-05T11:18:43.400607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:43.400607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.02204","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:18:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"R+pchKzXEBNv4/bjN2wF+u6Tn0qcA6Gq6Tte3vjM9qGlTWMpu5NWGCefpqDUr42+dMflKhwcbll6ydiDbTvtCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T05:17:10.488714Z"},"content_sha256":"bfc3cf200cae1ce24d20b2e37ce796ea25e9b8c5d177c526acfb18a0ee0452c4","schema_version":"1.0","event_id":"sha256:bfc3cf200cae1ce24d20b2e37ce796ea25e9b8c5d177c526acfb18a0ee0452c4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:RH2WQF6QVY3IOSFI3TZUKOBXIS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"BehaviorBox: Automated Discovery of Fine-Grained Performance Differences Between Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Lindia Tjuatja","submitted_at":"2025-06-02T19:44:06Z","abstract_excerpt":"Language model evaluation is a daunting task: prompts are brittle, corpus-level perplexities are vague, and the choice of benchmarks are endless. Finding examples that show meaningful, generalizable differences between two LMs is crucial to understanding where one model succeeds and another fails. Can this process be done automatically? In this work, we propose methodology for automated comparison of language models that uses performance-aware contextual embeddings to find fine-grained features of text where one LM outperforms another. Our method, which we name BehaviorBox, extracts coherent f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02204","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02204/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:18:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"LltAgbjEumYiBlOmyMZgHAP2HaJihO+tud2tBj5k6l74+d7W7xvRPazYnAum6FzxLY4qIiawAnc1I11hsa07BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T05:17:10.489575Z"},"content_sha256":"4803c0e9229f43142223db213b04bfe35ff32af839f8063794ab2119eef5cd21","schema_version":"1.0","event_id":"sha256:4803c0e9229f43142223db213b04bfe35ff32af839f8063794ab2119eef5cd21"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/bundle.json","state_url":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T05:17:10Z","links":{"resolver":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS","bundle":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/bundle.json","state":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:RH2WQF6QVY3IOSFI3TZUKOBXIS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"04f60d0b2643e4d5ab3c3d532637037693ce87a99ba495afc79e803957e35114","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:44:06Z","title_canon_sha256":"a99b2b8f0417ceb5c99a559ab71a9d90b90aac5a3ea9325376f915df128d7069"},"schema_version":"1.0","source":{"id":"2506.02204","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.02204","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"arxiv_version","alias_value":"2506.02204v2","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02204","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"pith_short_12","alias_value":"RH2WQF6QVY3I","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"pith_short_16","alias_value":"RH2WQF6QVY3IOSFI","created_at":"2026-07-05T11:18:43Z"},{"alias_kind":"pith_short_8","alias_value":"RH2WQF6Q","created_at":"2026-07-05T11:18:43Z"}],"graph_snapshots":[{"event_id":"sha256:4803c0e9229f43142223db213b04bfe35ff32af839f8063794ab2119eef5cd21","target":"graph","created_at":"2026-07-05T11:18:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.02204/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language model evaluation is a daunting task: prompts are brittle, corpus-level perplexities are vague, and the choice of benchmarks are endless. Finding examples that show meaningful, generalizable differences between two LMs is crucial to understanding where one model succeeds and another fails. Can this process be done automatically? In this work, we propose methodology for automated comparison of language models that uses performance-aware contextual embeddings to find fine-grained features of text where one LM outperforms another. Our method, which we name BehaviorBox, extracts coherent f","authors_text":"Graham Neubig, Lindia Tjuatja","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:44:06Z","title":"BehaviorBox: Automated Discovery of Fine-Grained Performance Differences Between Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02204","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:bfc3cf200cae1ce24d20b2e37ce796ea25e9b8c5d177c526acfb18a0ee0452c4","target":"record","created_at":"2026-07-05T11:18:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"04f60d0b2643e4d5ab3c3d532637037693ce87a99ba495afc79e803957e35114","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:44:06Z","title_canon_sha256":"a99b2b8f0417ceb5c99a559ab71a9d90b90aac5a3ea9325376f915df128d7069"},"schema_version":"1.0","source":{"id":"2506.02204","kind":"arxiv","version":2}},"canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","first_computed_at":"2026-07-05T11:18:43.400607Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:18:43.400607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"2uLgnXXtQXzKkvaQuCw1EDbphKBASKMMvaHW4QTwatVugB7w9reBcuwUmfQDGm0DZdW/aRK3zYhKlIASFYI4DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:18:43.401074Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.02204","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:bfc3cf200cae1ce24d20b2e37ce796ea25e9b8c5d177c526acfb18a0ee0452c4","sha256:4803c0e9229f43142223db213b04bfe35ff32af839f8063794ab2119eef5cd21"],"state_sha256":"6f55d190f316a51df50d3d43b40ac06a6b0612150e256df9a687095c38352809"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ggcF794RUNoyhJVcJ6614OIUOoubMFIyXjv0FSg1X/xJJ4JMkE6vca29ayp6GaWNJfKypiW+ewe62427tAjcBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T05:17:10.495302Z","bundle_sha256":"226b17ea5dc3cde47278505e07f231e69e31873b4b354a74abda3bf9d8aeaf43"}}