<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
    <channel>
        <title>Deltafin runs Kimi K3, 2.8 trillion parameters and 1.56 terabytes of weights, on a 64GB MacBook</title>
        <link>https://stream.andersonr.net/videos/watch/6ac027f6-1b55-4873-95ec-bcc353320869</link>
        <description>Deltafin runs Kimi K3, 2.8 trillion parameters and 1.56 terabytes of weights, on a 64GB MacBook. The trick is that a mixture-of-experts model only touches a sliver of itself per token, so the attention spine lives on local disk and the 82,432 routed experts stay on Hugging Face's CDN, fetched one HTTP range request at a time into a growing cache. It's about a token per minute. The author calls it an existence proof, not a chat setup. #github #opensource https://github.com/gavamedia/deltafin</description>
        <lastBuildDate>Wed, 29 Jul 2026 12:42:20 GMT</lastBuildDate>
        <docs>https://validator.w3.org/feed/docs/rss2.html</docs>
        <generator>PeerTube - https://stream.andersonr.net</generator>
        <image>
            <title>Deltafin runs Kimi K3, 2.8 trillion parameters and 1.56 terabytes of weights, on a 64GB MacBook</title>
            <url>https://stream.andersonr.net/client/assets/images/icons/icon-1500x1500.png</url>
            <link>https://stream.andersonr.net/videos/watch/6ac027f6-1b55-4873-95ec-bcc353320869</link>
        </image>
        <copyright>All rights reserved, unless otherwise specified in the terms specified at https://stream.andersonr.net/about and potential licenses granted by each content's rightholder.</copyright>
        <atom:link href="https://stream.andersonr.net/feeds/video-comments.xml?videoId=6ac027f6-1b55-4873-95ec-bcc353320869" rel="self" type="application/rss+xml"/>
    </channel>
</rss>