{"authors":[{"id":null,"fullName":"Chatzi, I.","name":"I.","surname":"Chatzi","rank":1,"pid":null},{"id":null,"fullName":"Corvelo Benz, N.","name":"N.","surname":"Corvelo Benz","rank":2,"pid":null},{"id":null,"fullName":"Tsirtsis, E.","name":"E.","surname":"Tsirtsis","rank":3,"pid":null},{"id":null,"fullName":"Gomez Rodriguez, M. ; https://orcid.org/0000-0003-3930-1161","name":"M. Https Orcid Org -. -. -.","surname":"Gomez Rodriguez","rank":4,"pid":null}],"openAccessColor":null,"publiclyFunded":false,"eoscIfGuidelines":null,"type":"publication","language":{"code":"eng","label":"English"},"countries":null,"subjects":null,"mainTitle":"Canonical Autoregressive Generation","subTitle":null,"descriptions":["State of the art large language models are trained using large amounts of tokens derived from raw text using what is called a tokenizer. Crucially, the tokenizer determines the (token) vocabulary a model will use during inference as well as, in principle, the (token) language. This is because, while the token vocabulary may allow for different tokenizations of a string, the tokenizer always maps the string to only one of these tokenizations--the canonical tokenization. However, multiple lines of empirical evidence suggest that large language models do not always generate canonical token sequences, and this comes with several negative consequences. In this work, we first show that, to generate a canonical token sequence, a model needs to generate (partial) canonical token sequences at each step of the autoregressive generation process underpinning its functioning. Building upon this theoretical result, we introduce canonical sampling, a simple and efficient sampling method that precludes a given model from generating non-canonical token sequences. Further, we also show that, in comparison with standard sampling, the distribution of token sequences generated using canonical sampling is provably closer to the true distribution of token sequences used during training."],"publicationDate":"2025-06-06","publisher":null,"embargoEndDate":null,"sources":null,"formats":["application/pdf"],"contributors":null,"coverages":null,"bestAccessRight":{"code":"c_abf2","label":"OPEN","scheme":"http://vocabularies.coar-repositories.org/documentation/access_rights/"},"container":null,"documentationUrls":null,"codeRepositoryUrl":null,"programmingLanguage":null,"contactPeople":null,"contactGroups":null,"tools":null,"size":null,"version":null,"geoLocations":null,"id":"od______1874::d7b07d047b0c7ea99b06e4de6e470c97","originalIds":["oai:pure.mpg.de:item_3667703","50|od______1874::d7b07d047b0c7ea99b06e4de6e470c97"],"pids":[{"scheme":"handle","value":"21.11116/0000-0011-B3CF-A"},{"scheme":"handle","value":"21.11116/0000-0011-B3D1-6"}],"dateOfCollection":null,"lastUpdateTimeStamp":null,"indicators":{"citationImpact":{"citationCount":0.0,"influence":2.1746283E-9,"popularity":2.2497804E-9,"impulse":0.0,"citationClass":"C5","influenceClass":"C5","impulseClass":"C5","popularityClass":"C5"}},"projects":[{"id":"corda__h2020::83d6aa753c4b6e94582f4ef733616892","code":"945719","acronym":"HumanML","title":"Human-Centric Machine Learning","funder":"European Commission","pids":[{"scheme":"doi","value":"10.3030/945719"}]}],"organizations":[{"legalName":"EIDGENOESSISCHE TECHNISCHE HOCHSCHULE ZUERICH","acronym":"ETH Zürich","id":"pending_org_::ee523ac5b92735484e281cd08b3b82f4","pids":[{"scheme":"PIC","value":"999979015"}],"countries":[{"code":"CH","label":"Switzerland"}],"websiteurl":"http://www.ethz.ch"},{"legalName":"Max Planck Institute for Software Systems","acronym":"MPI-SWS","id":"openorgs____::a6be3b23fdb0e61554b444ec8c1e8d1c","pids":[{"scheme":"Wikidata","value":"Q881238"},{"scheme":"GRID","value":"grid.469860.5"},{"scheme":"ISNI","value":"000000040492020X"},{"scheme":"ROR","value":"https://ror.org/02pe2kf23"},{"scheme":"wikidata","value":"Q881238"}],"countries":[{"code":"DE","label":"Germany"}],"websiteurl":"https://www.mpi-sws.org/index.php"},{"legalName":"ETH Zurich","acronym":"ETHZ","id":"openorgs____::fb1e14f93f04d43e1a10a9f17d12c669","pids":[{"scheme":"wikidata","value":"Q11942"},{"scheme":"FundRef","value":"501100003070"},{"scheme":"OrgRef","value":"210910"},{"scheme":"ROR","value":"https://ror.org/05a28rw58"},{"scheme":"fundref","value":"501100003006"},{"scheme":"FundRef","value":"501100003006"},{"scheme":"FundRef","value":"501100001710"},{"scheme":"fundref","value":"501100001710"},{"scheme":"GRID","value":"grid.5801.c"},{"scheme":"PIC","value":"999979015"},{"scheme":"Wikidata","value":"Q11942"},{"scheme":"fundref","value":"501100003070"},{"scheme":"mag_id","value":"35440088"},{"scheme":"OrgReg","value":"CH0012"},{"scheme":"RRID","value":"RRID:SCR_000962"},{"scheme":"RRID","value":"RRID:nlx_143698"},{"scheme":"ISNI","value":"0000000121562780"}],"countries":[{"code":"CH","label":"Switzerland"}],"websiteurl":"https://www.ethz.ch/en.html"},{"legalName":"Max Planck Society","acronym":"MPG","id":"openorgs____::5a405a89387d1881afa956f475994e10","pids":[{"scheme":"ISNI","value":"0000000121051091"},{"scheme":"ROR","value":"https://ror.org/01hhn8329"},{"scheme":"GRID","value":"grid.4372.2"},{"scheme":"PIC","value":"999990267"},{"scheme":"RRID","value":"RRID:nlx_55250"},{"scheme":"FundRef","value":"501100004189"},{"scheme":"RRID","value":"RRID:SCR_011366"},{"scheme":"Wikidata","value":"Q158085"},{"scheme":"OrgRef","value":"614753"},{"scheme":"mag_id","value":"149899117"},{"scheme":"wikidata","value":"Q158085"},{"scheme":"fundref","value":"501100004189"}],"countries":[{"code":"DE","label":"Germany"}],"websiteurl":"http://www.mpg.de/en"},{"legalName":"Department of Biosystems Science and Engineering (D-BSSE)","acronym":"Department of Biosystems Science and Engineering (D-BSSE)","id":"pending_org_::af83f38c07afd8d01d537bee79d55932","pids":null,"countries":[{"code":"CH","label":"Switzerland"}],"websiteurl":null}],"communities":null,"collectedFrom":[{"key":"opendoar____::958adb57686c2fdec5796398de5f317a","value":"MPG.PuRe"}],"instances":[{"pids":[{"scheme":"handle","value":"21.11116/0000-0011-B3CF-A"},{"scheme":"handle","value":"21.11116/0000-0011-B3D1-6"}],"alternateIdentifiers":[{"scheme":"doi","value":"10.48550/arxiv.2506.06446"},{"scheme":"arXiv","value":"2506.06446"}],"license":"CC BY","accessRight":{"code":"c_abf2","label":"OPEN","scheme":"http://vocabularies.coar-repositories.org/documentation/access_rights/","openAccessRoute":null},"type":"Research","urls":["https://hdl.handle.net/21.11116/0000-0011-B3CF-A","https://hdl.handle.net/21.11116/0000-0011-B3D1-6"],"publicationDate":"2025-06-06","refereed":"nonPeerReviewed","hostedBy":{"key":"opendoar____::958adb57686c2fdec5796398de5f317a","value":"MPG.PuRe"},"collectedFrom":{"key":"opendoar____::958adb57686c2fdec5796398de5f317a","value":"MPG.PuRe"}}],"links":[{"header":{"relationType":"resultProject","relationClass":"isProducedBy","relatedIdentifier":"corda__h2020::83d6aa753c4b6e94582f4ef733616892","relatedRecordType":"project","relationProvenance":"sysimport:crosswalk:repository","trust":"0.9"},"collectedfrom":[{"dsId":"openaire____::a55eb91348674d853191f4f4fd73d078","dsName":"CORDA - COmmon Research DAta Warehouse - Horizon 2020"}],"projectTitle":"Human-Centric Machine Learning","code":"945719","funding":{"funder":{"id":"ec__________::EC","shortname":"EC","name":"European Commission","jurisdiction":{"code":"EU","label":"European Union"},"pid":null},"level0":{"id":"ec__________::EC::H2020","description":"Horizon 2020 Framework Programme","name":"H2020"},"level1":{"id":"ec__________::EC::H2020::ERC","description":"European Research Council","name":"ERC"},"level2":{"id":"ec__________::EC::H2020::ERC::ERC-STG","description":"Starting Grant","name":"ERC-STG"}},"startDate":"2020-11-01","endDate":"2025-10-31"}],"otherTitles":null,"green":true,"inDiamondJournal":false,"isGreen":true,"isInDiamondJournal":false}