
		<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
			<channel>
				<title>José David Baena – Distributed Systems Engineer</title>
				<link>https://josedavidbaena.com</link>
				<description>Production notes and source-backed analysis on distributed systems, messaging infrastructure, open-source internals, and model engineering.</description>
				<language>en-us</language>
				<managingEditor>josedab@gmail.com (José David Baena)</managingEditor>
				<webMaster>josedab@gmail.com (José David Baena)</webMaster>
				<lastBuildDate>Fri, 19 Sep 2025 00:00:00 GMT</lastBuildDate>
				<atom:link href="https://josedavidbaena.com/tags/transformers/feed.xml" rel="self" type="application/rss+xml"/>
				
		<item>
			<guid>https://josedavidbaena.com/blog/tiny-language-models/efficient-attention-mechanisms-tiny-models</guid>
			<title>Efficient Attention Mechanisms for Tiny Language Models</title>
			<link>https://josedavidbaena.com/blog/tiny-language-models/efficient-attention-mechanisms-tiny-models</link>
			<description>Attention burns 50% of inference time and 75% of memory. MQA shrinks KV cache 4×. Flash Attention fuses kernels for 2–4× speedup. GQA splits the gap.</description>
			<pubDate>Fri, 19 Sep 2025 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>machine-learning</category><category>attention</category><category>transformers</category><category>optimization</category><category>tutorial</category>
		</item>
	
			</channel>
		</rss>
	