{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SLSMYQXKFJ6JYJZAYCDU7OO5C7","short_pith_number":"pith:SLSMYQXK","schema_version":"1.0","canonical_sha256":"92e4cc42ea2a7c9c2720c0874fb9dd17ed4d00069b8dcf560ff858a10e2cca93","source":{"kind":"arxiv","id":"2401.08399","version":2},"attestation_state":"computed","paper":{"title":"TACO: Benchmarking Generalizable Bimanual Tool-ACtion-Object Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haolin Yang, Ling Liu, Li Yi, Xu Si, Yebin Liu, Yun Liu, Yuxiang Zhang, Zipeng Li","submitted_at":"2024-01-16T14:41:42Z","abstract_excerpt":"Humans commonly work with multiple objects in daily life and can intuitively transfer manipulation skills to novel objects by understanding object functional regularities. However, existing technical approaches for analyzing and synthesizing hand-object manipulation are mostly limited to handling a single hand and object due to the lack of data support. To address this, we construct TACO, an extensive bimanual hand-object-interaction dataset spanning a large variety of tool-action-object compositions for daily human activities. TACO contains 2.5K motion sequences paired with third-person and e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08399","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-16T14:41:42Z","cross_cats_sorted":[],"title_canon_sha256":"5e5c1d32294f8041f12a8a07c5eaba9d886c968f8785f9ceb98267829f6e3031","abstract_canon_sha256":"0c9f68bb69553ca226701569cd5b4f9984dba59bfd8dc4ce321670d498100d0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:00.159488Z","signature_b64":"3xyhqEjksEMemrX6vgBW0kVsrKBIIM6kfGKa0j4ESL2MbsZ91+oikzhaNb2vtEPRJ8VzROJV7M224V5RxLIrAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92e4cc42ea2a7c9c2720c0874fb9dd17ed4d00069b8dcf560ff858a10e2cca93","last_reissued_at":"2026-07-05T08:00:00.158941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:00.158941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TACO: Benchmarking Generalizable Bimanual Tool-ACtion-Object Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haolin Yang, Ling Liu, Li Yi, Xu Si, Yebin Liu, Yun Liu, Yuxiang Zhang, Zipeng Li","submitted_at":"2024-01-16T14:41:42Z","abstract_excerpt":"Humans commonly work with multiple objects in daily life and can intuitively transfer manipulation skills to novel objects by understanding object functional regularities. However, existing technical approaches for analyzing and synthesizing hand-object manipulation are mostly limited to handling a single hand and object due to the lack of data support. To address this, we construct TACO, an extensive bimanual hand-object-interaction dataset spanning a large variety of tool-action-object compositions for daily human activities. TACO contains 2.5K motion sequences paired with third-person and e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08399","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08399/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08399","created_at":"2026-07-05T08:00:00.159007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08399v2","created_at":"2026-07-05T08:00:00.159007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08399","created_at":"2026-07-05T08:00:00.159007+00:00"},{"alias_kind":"pith_short_12","alias_value":"SLSMYQXKFJ6J","created_at":"2026-07-05T08:00:00.159007+00:00"},{"alias_kind":"pith_short_16","alias_value":"SLSMYQXKFJ6JYJZA","created_at":"2026-07-05T08:00:00.159007+00:00"},{"alias_kind":"pith_short_8","alias_value":"SLSMYQXK","created_at":"2026-07-05T08:00:00.159007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12604","citing_title":"EgoEngine: From Egocentric Human Videos to High-Fidelity Dexterous Robot Demonstrations","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12440","citing_title":"EgoVLA: Learning Vision-Language-Action Models from Egocentric Human Videos","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":193,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7","json":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7.json","graph_json":"https://pith.science/api/pith-number/SLSMYQXKFJ6JYJZAYCDU7OO5C7/graph.json","events_json":"https://pith.science/api/pith-number/SLSMYQXKFJ6JYJZAYCDU7OO5C7/events.json","paper":"https://pith.science/paper/SLSMYQXK"},"agent_actions":{"view_html":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7","download_json":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7.json","view_paper":"https://pith.science/paper/SLSMYQXK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08399&json=true","fetch_graph":"https://pith.science/api/pith-number/SLSMYQXKFJ6JYJZAYCDU7OO5C7/graph.json","fetch_events":"https://pith.science/api/pith-number/SLSMYQXKFJ6JYJZAYCDU7OO5C7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7/action/storage_attestation","attest_author":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7/action/author_attestation","sign_citation":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7/action/citation_signature","submit_replication":"https://pith.science/pith/SLSMYQXKFJ6JYJZAYCDU7OO5C7/action/replication_record"}},"created_at":"2026-07-05T08:00:00.159007+00:00","updated_at":"2026-07-05T08:00:00.159007+00:00"}