<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
	<channel>
		<title>Benchmarking on NicoLabs</title>
		<link>https://blog.hellonico.info/tags/benchmarking/</link>
		<description>Recent content in Benchmarking on NicoLabs</description>
		<generator>Hugo</generator>
		<language>en</language>
		
		
		
		
			<lastBuildDate>Sat, 10 Oct 2026 02:00:00 +0900</lastBuildDate>
		
			<atom:link href="https://blog.hellonico.info/tags/benchmarking/index.xml" rel="self" type="application/rss+xml" />
			<item>
				<title>125 Billion Parameters in 32GB RAM: Benchmarking Sushi, Qwen, and Thinking Tokens on Apple M4</title>
				<link>https://blog.hellonico.info/posts/sushi-qwen-125b-thinking-arena/</link>
				<pubDate>Sat, 10 Oct 2026 02:00:00 +0900</pubDate>
				<guid>https://blog.hellonico.info/posts/sushi-qwen-125b-thinking-arena/</guid>
				<description>&lt;p&gt;Let’s pause for a moment to appreciate the sheer, glorious absurdity of local AI in late 2026.&lt;/p&gt;&#xA;&lt;p&gt;Just three years ago, if you wanted to run a model with over 100 billion parameters, you needed a server chassis the size of a mini-fridge, an electrical circuit that could power a laundromat, and a bank loan to pay for four NVIDIA A100 GPUs. If someone told you that you would soon run a &lt;strong&gt;125-billion parameter reasoning model on a standard 32GB consumer Mac&lt;/strong&gt;, you would have politely recommended they seek medical attention.&lt;/p&gt;</description>
			</item>
	</channel>
</rss>
