@inproceedings{70b2cf048f014dc28733f9a92a9bc4b5,
title = "Large language model tokens are psychologically salient",
abstract = "Large language models segment words into chunks called tokens, using compression algorithms that ignore semantics. We investigated whether tokenization corrupts representations of word meanings in 17 languages. We found that GPT-4o and Llama 3 inflate the similarity of words that share tokens. However, tokens turned out to be good predictors of orthographic priming, such that people recognize a target word faster after reading a prime that ends with the same token. This boost in priming far exceeds what other overlapping strings of letters explain, which suggests that tokenization selectively identifies functional subword units. The pattern extends to the production of word associates in English: Tokens capture phonologically motivated associations, while other strings of letters do not. So, tokenization does influence semantic representations, but because tokens correspond to psychologically salient orthographic and/or phonological constituents, they may endow large language models with human-like language networks and facilitate alignment with human word processing. {\textcopyright}2025 the author(s).",
keywords = "large language models, tokenization, conceptual alignment, semantic priming, subword processing",
author = "Haslett, \{David A.\} and Chan, \{Antoni B.\} and Hsiao, \{Janet H.\}",
year = "2025",
month = jul,
language = "English",
series = "Proceedings of the Annual Meeting of the Cognitive Science Society",
publisher = "University of California",
pages = "4819--4827",
editor = "D. Barner and N.R. Bramley and A. Ruggeri and C.M. Walker",
booktitle = "Proceedings of the 47th Annual Conference of the Cognitive Science Society",
address = "United States",
note = "47th Annual Meeting of the Cognitive Science Society (CogSci 2025) ; Conference date: 30-07-2025 Through 02-08-2025",
}