
		<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
			<channel>
				<title>José David Baena – Software Engineer</title>
				<link>https://josedavidbaena.com</link>
				<description>Personal website and blog where I document thoughts, ideas, and interests in software engineering, web performance, and open source technologies.</description>
				<language>en-us</language>
				<managingEditor>josedab@gmail.com (José David Baena)</managingEditor>
				<webMaster>josedab@gmail.com (José David Baena)</webMaster>
				<lastBuildDate>Mon, 24 Nov 2025 00:00:00 GMT</lastBuildDate>
				<atom:link href="https://josedavidbaena.com/tags/bpe/feed.xml" rel="self" type="application/rss+xml"/>
				
		<item>
			<guid>https://josedavidbaena.com/blog/nanochat/tokenizer-design-choices-bpe-vocabulary</guid>
			<title>Tokenizer Design Choices: BPE, Vocabulary, and Implementation</title>
			<link>https://josedavidbaena.com/blog/nanochat/tokenizer-design-choices-bpe-vocabulary</link>
			<description>Your tokenizer decides what your model sees. 32K BPE vocab. 3.5 tokens per word. 10M tokens/sec in Rust. These choices compound at every training step.</description>
			<pubDate>Mon, 24 Nov 2025 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>nanochat</category><category>tokenization</category><category>bpe</category><category>tiktoken</category><category>practical-guide</category>
		</item>
	
			</channel>
		</rss>
	