@misc{zhang_emergent_2026,
 abstract = {Human observers prioritize visual information according to task goals. Most computational models of naturalistic viewing are gaze-trained for free viewing, leaving open whether goal-directed attention can emerge in systems without gaze supervision. We tested two off-the-shelf vision-language models (VLMs), Qwen3-VL-32B-Thinking and Gemma-4-26B-A4B-it, on 4,887 naturalistic scenes under visual-search and free-viewing instructions. Model predictions were compared with human fixations on the same images under corresponding tasks. Both models aligned more closely with human fixations under matching goals than under mismatched goals. This crossover persisted in target-absent scenes, where alignment could not be explained by simple visual grounding, and appeared in decoder-layer readouts. Furthermore, model-thinking traces were grounded in target semantics during search and in visual prominence during free viewing. These findings show that general-purpose VLMs can generate human-aligned, goal-directed spatial priorities without gaze-specific training, informing theories of goal-directed attention and offering scalable tools for predicting where people look across tasks.},
 author = {Zhang, Han},
 date = {2026-08-31},
 doi = {10.48550/arXiv.2609.05517},
 eprint = {2609.05517 [cs.CV]},
 eprinttype = {arxiv},
 file = {Preprint PDF:/Users/hanzh/Zotero/storage/BNHRSR8J/Zhang - 2026 - Emergent goal-directed attention in large vision-language models.pdf:application/pdf;Snapshot:/Users/hanzh/Zotero/storage/CK33794E/2609.html:text/html},
 keywords = {Computer Science - Artificial Intelligence, Computer Science - Computation and Language, Computer Science - Computer Vision and Pattern Recognition},
 publisher = {arXiv},
 title = {Emergent goal-directed attention in large vision-language models},
 urldate = {2026-09-09}
}
