<?xml version="1.0" encoding="UTF-8"?>
<feed xmlns="http://www.w3.org/2005/Atom">
    <title>tpu</title>
    <link rel="self" type="application/atom+xml" href="https://links.biapy.com/guest/tags/1243/feed"/>
    <updated>2026-07-26T22:17:26+00:00</updated>
    <id>https://links.biapy.com/guest/tags/1243/feed</id>
            <entry>
            <id>https://links.biapy.com/links/1677</id>
            <title type="text"><![CDATA[How To Scale Your Model]]></title>
            <link rel="alternate" href="https://jax-ml.github.io/scaling-book/" />
            <link rel="via" type="application/atom+xml" href="https://links.biapy.com/links/1677"/>
            <author>
                <name><![CDATA[Biapy]]></name>
            </author>
            <summary type="text">
                <![CDATA[A Systems View of LLMs on TPUs.

This book aims to demystify the art of scaling LLMs on TPUs. We try to explain how TPUs work, how LLMs actually run at scale, and how to pick parallelism schemes during training and inference that avoid communication bottlenecks.

- [How To Scale Your Model @ GitHub](https://github.com/jax-ml/scaling-book/).]]>
            </summary>
            <updated>2025-08-28T20:35:47+00:00</updated>
        </entry>
    </feed>
