{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TQVPIUIUMXSEBVGQCJ5AH5PTIC","short_pith_number":"pith:TQVPIUIU","schema_version":"1.0","canonical_sha256":"9c2af4511465e440d4d0127a03f5f34084262e484bc882fc818d31020fd55e92","source":{"kind":"arxiv","id":"2509.09321","version":1},"attestation_state":"computed","paper":{"title":"Towards Adaptive ML Benchmarks: Web-Agent-Driven Construction, Domain Expansion, and Metric Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Feng Wei, Hangyi Jia, Hanwen Tong, Lin Chen, Xinhui Wu, YuXi Qian","submitted_at":"2025-09-11T10:10:48Z","abstract_excerpt":"Recent advances in large language models (LLMs) have enabled the emergence of general-purpose agents for automating end-to-end machine learning (ML) workflows, including data analysis, feature engineering, model training, and competition solving. However, existing benchmarks remain limited in task coverage, domain diversity, difficulty modeling, and evaluation rigor, failing to capture the full capabilities of such agents in realistic settings. We present TAM Bench, a diverse, realistic, and structured benchmark for evaluating LLM-based agents on end-to-end ML tasks. TAM Bench features three k"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.09321","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-09-11T10:10:48Z","cross_cats_sorted":[],"title_canon_sha256":"aaf051dee893b6016bb75d8b6b7daef27cb727b956d58f1492e5aba522fd8d0a","abstract_canon_sha256":"680061c239f3c9d0088a2f6e09d40d603a11d8ad9c39ec3297fde6f9b2acb53f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:33.885939Z","signature_b64":"tsj6Q/WSWN/lPJBp2xG6uPKk3HRMAoF8gE3x/7Qb1CXLTlz0wm5qWaU3Wiv7e2KWUYPWtXhdov1MHQG8nZclAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9c2af4511465e440d4d0127a03f5f34084262e484bc882fc818d31020fd55e92","last_reissued_at":"2026-07-05T12:09:33.885416Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:33.885416Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Adaptive ML Benchmarks: Web-Agent-Driven Construction, Domain Expansion, and Metric Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Feng Wei, Hangyi Jia, Hanwen Tong, Lin Chen, Xinhui Wu, YuXi Qian","submitted_at":"2025-09-11T10:10:48Z","abstract_excerpt":"Recent advances in large language models (LLMs) have enabled the emergence of general-purpose agents for automating end-to-end machine learning (ML) workflows, including data analysis, feature engineering, model training, and competition solving. However, existing benchmarks remain limited in task coverage, domain diversity, difficulty modeling, and evaluation rigor, failing to capture the full capabilities of such agents in realistic settings. We present TAM Bench, a diverse, realistic, and structured benchmark for evaluating LLM-based agents on end-to-end ML tasks. TAM Bench features three k"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09321","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.09321/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.09321","created_at":"2026-07-05T12:09:33.885483+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.09321v1","created_at":"2026-07-05T12:09:33.885483+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09321","created_at":"2026-07-05T12:09:33.885483+00:00"},{"alias_kind":"pith_short_12","alias_value":"TQVPIUIUMXSE","created_at":"2026-07-05T12:09:33.885483+00:00"},{"alias_kind":"pith_short_16","alias_value":"TQVPIUIUMXSEBVGQ","created_at":"2026-07-05T12:09:33.885483+00:00"},{"alias_kind":"pith_short_8","alias_value":"TQVPIUIU","created_at":"2026-07-05T12:09:33.885483+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20683","citing_title":"From Question Answering to Task Completion: A Survey on Agent System and Harness Design","ref_index":89,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC","json":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC.json","graph_json":"https://pith.science/api/pith-number/TQVPIUIUMXSEBVGQCJ5AH5PTIC/graph.json","events_json":"https://pith.science/api/pith-number/TQVPIUIUMXSEBVGQCJ5AH5PTIC/events.json","paper":"https://pith.science/paper/TQVPIUIU"},"agent_actions":{"view_html":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC","download_json":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC.json","view_paper":"https://pith.science/paper/TQVPIUIU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.09321&json=true","fetch_graph":"https://pith.science/api/pith-number/TQVPIUIUMXSEBVGQCJ5AH5PTIC/graph.json","fetch_events":"https://pith.science/api/pith-number/TQVPIUIUMXSEBVGQCJ5AH5PTIC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC/action/storage_attestation","attest_author":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC/action/author_attestation","sign_citation":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC/action/citation_signature","submit_replication":"https://pith.science/pith/TQVPIUIUMXSEBVGQCJ5AH5PTIC/action/replication_record"}},"created_at":"2026-07-05T12:09:33.885483+00:00","updated_at":"2026-07-05T12:09:33.885483+00:00"}