
		<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
			<channel>
				<title>José David Baena – Software Engineer</title>
				<link>https://josedavidbaena.com</link>
				<description>Personal website and blog where I document thoughts, ideas, and interests in software engineering, web performance, and open source technologies.</description>
				<language>en-us</language>
				<managingEditor>josedab@gmail.com (José David Baena)</managingEditor>
				<webMaster>josedab@gmail.com (José David Baena)</webMaster>
				<lastBuildDate>Sun, 26 Jul 2026 00:00:00 GMT</lastBuildDate>
				<atom:link href="https://josedavidbaena.com/tags/gpu-topology/feed.xml" rel="self" type="application/rss+xml"/>
				
		<item>
			<guid>https://josedavidbaena.com/blog/frontier-model-engineering/04-multi-gpu-inference-first-principles</guid>
			<title>Multi-GPU Inference Starts With Per-Rank Placement</title>
			<link>https://josedavidbaena.com/blog/frontier-model-engineering/04-multi-gpu-inference-first-principles</link>
			<description>Prove every rank&#39;s weight and KV placement, map collectives onto physical links, pin runtime behavior, benchmark the workload, and plan rank recovery.</description>
			<pubDate>Sun, 26 Jul 2026 00:00:00 GMT</pubDate>
			<author>josedab@gmail.com (José David Baena)</author>
			<category>multi-gpu</category><category>distributed-inference</category><category>tensor-parallelism</category><category>expert-parallelism</category><category>gpu-topology</category>
		</item>
	
			</channel>
		</rss>
	