{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y6ULJ4GQYVAW2AI4NP5LSFWOQC","short_pith_number":"pith:Y6ULJ4GQ","schema_version":"1.0","canonical_sha256":"c7a8b4f0d0c5416d011c6bfab916ce809172896f3cc9aefea3cca4aac9468fa3","source":{"kind":"arxiv","id":"2412.15106","version":1},"attestation_state":"computed","paper":{"title":"Knowing Where to Focus: Attention-Guided Alignment for Text-based Person Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jie Chen, Lei Tan, Liujuan Cao, Pingyang Dai, Rongrong Ji, Weihao Li","submitted_at":"2024-12-19T17:51:49Z","abstract_excerpt":"In the realm of Text-Based Person Search (TBPS), mainstream methods aim to explore more efficient interaction frameworks between text descriptions and visual data. However, recent approaches encounter two principal challenges. Firstly, the widely used random-based Masked Language Modeling (MLM) considers all the words in the text equally during training. However, massive semantically vacuous words ('with', 'the', etc.) be masked fail to contribute efficient interaction in the cross-modal MLM and hampers the representation alignment. Secondly, manual descriptions in TBPS datasets are tedious an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15106","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-19T17:51:49Z","cross_cats_sorted":[],"title_canon_sha256":"b9d9687cd1563df722b3292dff34d686364ec592bcf226d7ca6162a6b110bb76","abstract_canon_sha256":"60fbec581c9665c18d79c833a0ac5eb6f8c30e508aa2dbf9cb76899c36c04a0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:57.972063Z","signature_b64":"I5WgHKMrjOtedUGHnIzYN9IFavSlAtC69BvGWKXPFquZdlgQQsS2oZ+8S3SHmc76UT7yQ15wyQ09ygzSHlr7Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7a8b4f0d0c5416d011c6bfab916ce809172896f3cc9aefea3cca4aac9468fa3","last_reissued_at":"2026-07-05T09:51:57.971614Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:57.971614Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Knowing Where to Focus: Attention-Guided Alignment for Text-based Person Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jie Chen, Lei Tan, Liujuan Cao, Pingyang Dai, Rongrong Ji, Weihao Li","submitted_at":"2024-12-19T17:51:49Z","abstract_excerpt":"In the realm of Text-Based Person Search (TBPS), mainstream methods aim to explore more efficient interaction frameworks between text descriptions and visual data. However, recent approaches encounter two principal challenges. Firstly, the widely used random-based Masked Language Modeling (MLM) considers all the words in the text equally during training. However, massive semantically vacuous words ('with', 'the', etc.) be masked fail to contribute efficient interaction in the cross-modal MLM and hampers the representation alignment. Secondly, manual descriptions in TBPS datasets are tedious an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15106","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15106/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15106","created_at":"2026-07-05T09:51:57.971671+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15106v1","created_at":"2026-07-05T09:51:57.971671+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15106","created_at":"2026-07-05T09:51:57.971671+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y6ULJ4GQYVAW","created_at":"2026-07-05T09:51:57.971671+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y6ULJ4GQYVAW2AI4","created_at":"2026-07-05T09:51:57.971671+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y6ULJ4GQ","created_at":"2026-07-05T09:51:57.971671+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01725","citing_title":"Motion-Aware Caching for Efficient Autoregressive Video Generation","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC","json":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC.json","graph_json":"https://pith.science/api/pith-number/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/graph.json","events_json":"https://pith.science/api/pith-number/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/events.json","paper":"https://pith.science/paper/Y6ULJ4GQ"},"agent_actions":{"view_html":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC","download_json":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC.json","view_paper":"https://pith.science/paper/Y6ULJ4GQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15106&json=true","fetch_graph":"https://pith.science/api/pith-number/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/graph.json","fetch_events":"https://pith.science/api/pith-number/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/action/storage_attestation","attest_author":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/action/author_attestation","sign_citation":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/action/citation_signature","submit_replication":"https://pith.science/pith/Y6ULJ4GQYVAW2AI4NP5LSFWOQC/action/replication_record"}},"created_at":"2026-07-05T09:51:57.971671+00:00","updated_at":"2026-07-05T09:51:57.971671+00:00"}