
		<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
			<channel>
				<title>José David Baena – Software Engineer</title>
				<link>https://josedavidbaena.com</link>
				<description>Personal website and blog where I document thoughts, ideas, and interests in software engineering, web performance, and open source technologies.</description>
				<language>en-us</language>
				<managingEditor>josedab@gmail.com (José David Baena)</managingEditor>
				<webMaster>josedab@gmail.com (José David Baena)</webMaster>
				<lastBuildDate>Sat, 22 Nov 2025 00:00:00 GMT</lastBuildDate>
				<atom:link href="https://josedavidbaena.com/tags/benchmarks/feed.xml" rel="self" type="application/rss+xml"/>
				
		<item>
			<guid>https://josedavidbaena.com/blog/nanochat/building-custom-evaluation-tasks</guid>
			<title>Building Custom Evaluation Tasks</title>
			<link>https://josedavidbaena.com/blog/nanochat/building-custom-evaluation-tasks</link>
			<description>Standard benchmarks measure general capabilities. Custom tasks measure what you actually care about. Build any domain benchmark in 50 lines of code.</description>
			<pubDate>Sat, 22 Nov 2025 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>nanochat</category><category>evaluation</category><category>benchmarks</category><category>core</category><category>testing</category><category>practical-guide</category>
		</item>
	
		<item>
			<guid>https://josedavidbaena.com/blog/tiny-language-models/tiny-llm-case-studies-production</guid>
			<title>Tiny LLM Deployment Patterns: Architecture Blueprints from Published Benchmarks</title>
			<link>https://josedavidbaena.com/blog/tiny-language-models/tiny-llm-case-studies-production</link>
			<description>Deployment patterns for tiny LLMs in healthcare, legal, manufacturing, and edge—grounded in published benchmarks from Microsoft, Apple, and MLPerf.</description>
			<pubDate>Mon, 13 Oct 2025 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>machine-learning</category><category>deployment-patterns</category><category>production</category><category>benchmarks</category><category>architecture</category>
		</item>
	
		<item>
			<guid>https://josedavidbaena.com/blog/tiny-language-models/tiny-llm-architecture-comparison</guid>
			<title>Tiny LLM Architecture Comparison: TinyLlama vs Phi-2 vs Gemma vs MobileLLM</title>
			<link>https://josedavidbaena.com/blog/tiny-language-models/tiny-llm-architecture-comparison</link>
			<description>Seven tiny models, one decision. Phi-2 wins on reasoning (56.7% MMLU). MobileLLM on speed (120 tok/s). Qwen on multilingual. Match constraints to model.</description>
			<pubDate>Sat, 20 Sep 2025 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>machine-learning</category><category>llm</category><category>architecture</category><category>benchmarks</category><category>comparison</category>
		</item>
	
			</channel>
		</rss>
	