<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
    <channel>
        <title>Swiftlet: streams 80B Qwen MoE experts from iPhone SSD, 2.5GB RAM, one token per second</title>
        <link>https://stream.andersonr.net/videos/watch/c35fdc80-5213-4ae5-ad7f-240ba9818568</link>
        <description>Swiftlet runs 35B and 80B Qwen mixture-of-experts models on Apple devices by keeping the dense core in memory and streaming routed experts from SSD. QPack files turn each expert fetch into one read, while caching and Metal kernels handle inference. The maintainer reports the 35B model using about 2.5 gigabytes of RAM on an iPhone 17 at roughly one token per second. #github #opensource https://github.com/leonickson1/Swiftlet</description>
        <lastBuildDate>Tue, 01 Sep 2026 11:59:40 GMT</lastBuildDate>
        <docs>https://validator.w3.org/feed/docs/rss2.html</docs>
        <generator>PeerTube - https://stream.andersonr.net</generator>
        <image>
            <title>Swiftlet: streams 80B Qwen MoE experts from iPhone SSD, 2.5GB RAM, one token per second</title>
            <url>https://stream.andersonr.net/client/assets/images/icons/icon-1500x1500.png</url>
            <link>https://stream.andersonr.net/videos/watch/c35fdc80-5213-4ae5-ad7f-240ba9818568</link>
        </image>
        <copyright>All rights reserved, unless otherwise specified in the terms specified at https://stream.andersonr.net/about and potential licenses granted by each content's rightholder.</copyright>
        <atom:link href="https://stream.andersonr.net/feeds/video-comments.xml?videoId=c35fdc80-5213-4ae5-ad7f-240ba9818568" rel="self" type="application/rss+xml"/>
    </channel>
</rss>