{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RH2WQF6QVY3IOSFI3TZUKOBXIS","short_pith_number":"pith:RH2WQF6Q","schema_version":"1.0","canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","source":{"kind":"arxiv","id":"2506.02204","version":2},"attestation_state":"computed","paper":{"title":"BehaviorBox: Automated Discovery of Fine-Grained Performance Differences Between Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Lindia Tjuatja","submitted_at":"2025-06-02T19:44:06Z","abstract_excerpt":"Language model evaluation is a daunting task: prompts are brittle, corpus-level perplexities are vague, and the choice of benchmarks are endless. Finding examples that show meaningful, generalizable differences between two LMs is crucial to understanding where one model succeeds and another fails. Can this process be done automatically? In this work, we propose methodology for automated comparison of language models that uses performance-aware contextual embeddings to find fine-grained features of text where one LM outperforms another. Our method, which we name BehaviorBox, extracts coherent f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02204","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-02T19:44:06Z","cross_cats_sorted":[],"title_canon_sha256":"a99b2b8f0417ceb5c99a559ab71a9d90b90aac5a3ea9325376f915df128d7069","abstract_canon_sha256":"04f60d0b2643e4d5ab3c3d532637037693ce87a99ba495afc79e803957e35114"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:43.401074Z","signature_b64":"2uLgnXXtQXzKkvaQuCw1EDbphKBASKMMvaHW4QTwatVugB7w9reBcuwUmfQDGm0DZdW/aRK3zYhKlIASFYI4DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89f56817d0ae368748a8dcf3453837449482ed4350b4ff7efa19c2de7363fdbc","last_reissued_at":"2026-07-05T11:18:43.400607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:43.400607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BehaviorBox: Automated Discovery of Fine-Grained Performance Differences Between Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Graham Neubig, Lindia Tjuatja","submitted_at":"2025-06-02T19:44:06Z","abstract_excerpt":"Language model evaluation is a daunting task: prompts are brittle, corpus-level perplexities are vague, and the choice of benchmarks are endless. Finding examples that show meaningful, generalizable differences between two LMs is crucial to understanding where one model succeeds and another fails. Can this process be done automatically? In this work, we propose methodology for automated comparison of language models that uses performance-aware contextual embeddings to find fine-grained features of text where one LM outperforms another. Our method, which we name BehaviorBox, extracts coherent f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02204","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02204/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02204","created_at":"2026-07-05T11:18:43.400660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02204v2","created_at":"2026-07-05T11:18:43.400660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02204","created_at":"2026-07-05T11:18:43.400660+00:00"},{"alias_kind":"pith_short_12","alias_value":"RH2WQF6QVY3I","created_at":"2026-07-05T11:18:43.400660+00:00"},{"alias_kind":"pith_short_16","alias_value":"RH2WQF6QVY3IOSFI","created_at":"2026-07-05T11:18:43.400660+00:00"},{"alias_kind":"pith_short_8","alias_value":"RH2WQF6Q","created_at":"2026-07-05T11:18:43.400660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS","json":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS.json","graph_json":"https://pith.science/api/pith-number/RH2WQF6QVY3IOSFI3TZUKOBXIS/graph.json","events_json":"https://pith.science/api/pith-number/RH2WQF6QVY3IOSFI3TZUKOBXIS/events.json","paper":"https://pith.science/paper/RH2WQF6Q"},"agent_actions":{"view_html":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS","download_json":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS.json","view_paper":"https://pith.science/paper/RH2WQF6Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02204&json=true","fetch_graph":"https://pith.science/api/pith-number/RH2WQF6QVY3IOSFI3TZUKOBXIS/graph.json","fetch_events":"https://pith.science/api/pith-number/RH2WQF6QVY3IOSFI3TZUKOBXIS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/action/storage_attestation","attest_author":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/action/author_attestation","sign_citation":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/action/citation_signature","submit_replication":"https://pith.science/pith/RH2WQF6QVY3IOSFI3TZUKOBXIS/action/replication_record"}},"created_at":"2026-07-05T11:18:43.400660+00:00","updated_at":"2026-07-05T11:18:43.400660+00:00"}