{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:KR442H2ES65M7V4HNQXIZQPINC","short_pith_number":"pith:KR442H2E","schema_version":"1.0","canonical_sha256":"5479cd1f4497bacfd7876c2e8cc1e86882fe8d8803503e8d8afc7e5cf30063e4","source":{"kind":"arxiv","id":"2105.09939","version":1},"attestation_state":"computed","paper":{"title":"Face, Body, Voice: Video Person-Clustering with Multiple Modalities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrew Brown, Andrew Zisserman, Vicky Kalogeiton","submitted_at":"2021-05-20T17:59:40Z","abstract_excerpt":"The objective of this work is person-clustering in videos -- grouping characters according to their identity. Previous methods focus on the narrower task of face-clustering, and for the most part ignore other cues such as the person's voice, their overall appearance (hair, clothes, posture), and the editing structure of the videos. Similarly, most current datasets evaluate only the task of face-clustering, rather than person-clustering. This limits their applicability to downstream applications such as story understanding which require person-level, rather than only face-level, reasoning. In t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.09939","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-05-20T17:59:40Z","cross_cats_sorted":[],"title_canon_sha256":"6bab0cda8a210c88f00585acf80300c3893672043a355b0b4c54b5c6aa3fe635","abstract_canon_sha256":"6778c64247a45002c7cb3df0f86842478b16fbb2a8b1b0fba02cf98699ae98b9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:41:58.939969Z","signature_b64":"Hle12m0Jysj+bZdaA0+CZ7Taxpn+qwHPCyGkwk0dXFSPzz7Qvvk9C/mTdvtSmpZYGFjlvKpYM9XsIi5GYTEEAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5479cd1f4497bacfd7876c2e8cc1e86882fe8d8803503e8d8afc7e5cf30063e4","last_reissued_at":"2026-07-05T02:41:58.939427Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:41:58.939427Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Face, Body, Voice: Video Person-Clustering with Multiple Modalities","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrew Brown, Andrew Zisserman, Vicky Kalogeiton","submitted_at":"2021-05-20T17:59:40Z","abstract_excerpt":"The objective of this work is person-clustering in videos -- grouping characters according to their identity. Previous methods focus on the narrower task of face-clustering, and for the most part ignore other cues such as the person's voice, their overall appearance (hair, clothes, posture), and the editing structure of the videos. Similarly, most current datasets evaluate only the task of face-clustering, rather than person-clustering. This limits their applicability to downstream applications such as story understanding which require person-level, rather than only face-level, reasoning. In t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.09939","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.09939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.09939","created_at":"2026-07-05T02:41:58.939485+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.09939v1","created_at":"2026-07-05T02:41:58.939485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.09939","created_at":"2026-07-05T02:41:58.939485+00:00"},{"alias_kind":"pith_short_12","alias_value":"KR442H2ES65M","created_at":"2026-07-05T02:41:58.939485+00:00"},{"alias_kind":"pith_short_16","alias_value":"KR442H2ES65M7V4H","created_at":"2026-07-05T02:41:58.939485+00:00"},{"alias_kind":"pith_short_8","alias_value":"KR442H2E","created_at":"2026-07-05T02:41:58.939485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC","json":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC.json","graph_json":"https://pith.science/api/pith-number/KR442H2ES65M7V4HNQXIZQPINC/graph.json","events_json":"https://pith.science/api/pith-number/KR442H2ES65M7V4HNQXIZQPINC/events.json","paper":"https://pith.science/paper/KR442H2E"},"agent_actions":{"view_html":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC","download_json":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC.json","view_paper":"https://pith.science/paper/KR442H2E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.09939&json=true","fetch_graph":"https://pith.science/api/pith-number/KR442H2ES65M7V4HNQXIZQPINC/graph.json","fetch_events":"https://pith.science/api/pith-number/KR442H2ES65M7V4HNQXIZQPINC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC/action/storage_attestation","attest_author":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC/action/author_attestation","sign_citation":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC/action/citation_signature","submit_replication":"https://pith.science/pith/KR442H2ES65M7V4HNQXIZQPINC/action/replication_record"}},"created_at":"2026-07-05T02:41:58.939485+00:00","updated_at":"2026-07-05T02:41:58.939485+00:00"}