<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
    <channel>
        <title>Qwen3.8-27B on one RTX 3090: 417 tokens a second from a 27B model on one 24GB gaming card</title>
        <link>https://stream.andersonr.net/videos/watch/faf01cfa-a042-4665-9320-9062556a21e5</link>
        <description>Fitting a 27-billion-parameter model onto one 24GB gaming card usually means stopping at "it loads." This recipe pushes past that to 417 tokens a second batched, and the wins are unglamorous in the best way. Qwen's untied embeddings ship as two unquantized 2.5GB matrices nobody bothered with, so requantizing both to int8 hands back 2.6GB, and on this architecture spare memory converts straight into batch size. #github #opensource https://github.com/syv-ai/qwen38-27b-rtx3090</description>
        <lastBuildDate>Tue, 01 Sep 2026 11:59:14 GMT</lastBuildDate>
        <docs>https://validator.w3.org/feed/docs/rss2.html</docs>
        <generator>PeerTube - https://stream.andersonr.net</generator>
        <image>
            <title>Qwen3.8-27B on one RTX 3090: 417 tokens a second from a 27B model on one 24GB gaming card</title>
            <url>https://stream.andersonr.net/client/assets/images/icons/icon-1500x1500.png</url>
            <link>https://stream.andersonr.net/videos/watch/faf01cfa-a042-4665-9320-9062556a21e5</link>
        </image>
        <copyright>All rights reserved, unless otherwise specified in the terms specified at https://stream.andersonr.net/about and potential licenses granted by each content's rightholder.</copyright>
        <atom:link href="https://stream.andersonr.net/feeds/video-comments.xml?videoId=faf01cfa-a042-4665-9320-9062556a21e5" rel="self" type="application/rss+xml"/>
    </channel>
</rss>