From 3ecb5888708839bcd36db079b5d4b62f6f7a53ce Mon Sep 17 00:00:00 2001 From: Dinesh <13635627+HumbleBee14@users.noreply.github.com> Date: Wed, 16 Sep 2026 12:08:59 -0700 Subject: [PATCH 1/4] Homepage refinements --- src/components/HeroFigure.tsx | 2 +- src/components/Playground.tsx | 2 + src/components/ToolGlyph.tsx | 16 + src/content/tools/attention-viz/index.mdx | 2 +- src/content/tools/eval-harness/index.mdx | 27 +- src/content/tools/throughput-calc/index.mdx | 2 +- .../tools/tokenizer-explorer/index.mdx | 19 +- .../TrainingComputeCalc.tsx | 315 ++++++++++++++++++ .../training-compute-calc/compute.test.ts | 80 +++++ .../tools/training-compute-calc/compute.ts | 52 +++ .../tools/training-compute-calc/index.mdx | 55 +++ src/pages/index.astro | 122 ++++--- src/styles/publication.css | 40 ++- 13 files changed, 616 insertions(+), 118 deletions(-) create mode 100644 src/content/tools/training-compute-calc/TrainingComputeCalc.tsx create mode 100644 src/content/tools/training-compute-calc/compute.test.ts create mode 100644 src/content/tools/training-compute-calc/compute.ts create mode 100644 src/content/tools/training-compute-calc/index.mdx diff --git a/src/components/HeroFigure.tsx b/src/components/HeroFigure.tsx index 21ae5fc..b0841ba 100644 --- a/src/components/HeroFigure.tsx +++ b/src/components/HeroFigure.tsx @@ -62,7 +62,7 @@ export function AttentionFig({ t }: { t: number }) { // Build the causal triangle in under two seconds, then scan the completed rows. const reveal = t / 0.14; const scan = t < 1.96 ? reveal : ((t - 1.96) % 2.8) / 0.2; - const activeRow = Math.min(N - 1, Math.floor(scan)); + const activeRow = Math.max(0, Math.min(N - 1, Math.floor(scan))); return ( = { 'throughput-calc': ThroughputCalc, 'gpu-mem-calc': GpuMemoryCalc, 'attention-viz': AttentionViz, + 'training-compute-calc': TrainingComputeCalc, }; export default function Playground({ tools }: { tools: PlaygroundTool[] }) { diff --git a/src/components/ToolGlyph.tsx b/src/components/ToolGlyph.tsx index 4318051..8bb0fb6 100644 --- a/src/components/ToolGlyph.tsx +++ b/src/components/ToolGlyph.tsx @@ -47,6 +47,22 @@ export default function ToolGlyph({ id }: { id: string }) { ); + case 'training-compute-calc': + return ( + + + + + + + ); case 'model-card': return ( diff --git a/src/content/tools/attention-viz/index.mdx b/src/content/tools/attention-viz/index.mdx index c95d282..3502a8c 100644 --- a/src/content/tools/attention-viz/index.mdx +++ b/src/content/tools/attention-viz/index.mdx @@ -1,6 +1,6 @@ --- name: Attention Visualizer -summary: Inspect attention patterns layer-by-layer for any Hugging Face model. Click any head to see its causal mask, induction behavior, and sink tokens. +summary: Explore a token-by-token attention map, from causal masking to recurring attention patterns. tag: Live icon: ◎ authors: diff --git a/src/content/tools/eval-harness/index.mdx b/src/content/tools/eval-harness/index.mdx index c930fc0..49ad003 100644 --- a/src/content/tools/eval-harness/index.mdx +++ b/src/content/tools/eval-harness/index.mdx @@ -1,33 +1,20 @@ --- name: Eval Harness Playground -summary: Run a small set of evaluations against any inference endpoint and get back a structured scorecard — quality, latency, cost, and refusal rate side by side. -tag: Beta +summary: Run a small, fixed set of evals against an OpenAI-compatible endpoint and get a scorecard for quality, latency, and cost. +tag: Soon icon: ▤ authors: - - lchen + - dinesh topics: - evaluation tags: [evals, benchmarks, scorecards] core: true --- -## What it does +## What we are building -Points at any inference endpoint (yours or a hosted one), runs a curated set of small but informative evals, and produces a scorecard. The set is small on purpose: you get answers in minutes instead of hours, and the evals are chosen for signal-to-noise rather than headline numbers. +Full benchmark suites such as lm-evaluation-harness take hours and real money to run. Most of the time the question is simpler: is this endpoint worth a closer look? This tool will point at any OpenAI-compatible endpoint, run a small curated set of prompts, and return a scorecard with quality, latency, and cost side by side. -## Why it's useful +## Status -Full benchmark suites take hours and cost real money. Most of the time you just want a quick "is this model worth a closer look." This playground gives you a focused first read so you can decide whether to invest in a full eval pass. - -## How to use it - -1. Provide an endpoint URL and auth header (or pick a hosted preset). -2. Select an eval pack — reasoning, code, instruction following, refusal behavior. -3. Run. Results stream in as each prompt completes. -4. Export the scorecard or share a permalink. - -## Limitations - -- Beta: the eval set is opinionated and will keep evolving. -- Not a replacement for full evaluation suites when stakes are high. -- Caches results by prompt hash, so re-running the same prompts is free. +Coming soon. Not yet available. diff --git a/src/content/tools/throughput-calc/index.mdx b/src/content/tools/throughput-calc/index.mdx index c8f25b4..fd4949d 100644 --- a/src/content/tools/throughput-calc/index.mdx +++ b/src/content/tools/throughput-calc/index.mdx @@ -1,6 +1,6 @@ --- name: Throughput Calculator -summary: Estimate tokens/sec for any model, precision, batch size, and GPU combination — memory-bound roofline only, no kernel-quality wishful thinking. +summary: Estimate memory-bound token throughput and KV-cache size across models, precisions, batch sizes, and GPUs. tag: Live icon: ◐ authors: diff --git a/src/content/tools/tokenizer-explorer/index.mdx b/src/content/tools/tokenizer-explorer/index.mdx index e057a36..1ff278e 100644 --- a/src/content/tools/tokenizer-explorer/index.mdx +++ b/src/content/tools/tokenizer-explorer/index.mdx @@ -1,27 +1,20 @@ --- name: Tokenizer explorer -summary: Paste any text and see how seven popular tokenizers split it — BPE, WordPiece, SentencePiece, tiktoken, and friends, side by side. -tag: Experimental +summary: Paste any text and see how popular tokenizers split it, side by side, with token counts and cost per provider. +tag: Soon icon: ▤ authors: - - lchen + - dinesh topics: - inference tags: [tokenizer, bpe, sentencepiece] featured: true --- -## What it does +## What we are building -Tokenizers determine how language models see your text. The same prompt can produce wildly different token counts (and therefore cost, latency, and behavior) across model families. This tool runs your text through seven tokenizers in parallel and shows the splits aligned column-by-column. - -## What you'll see - -- Token-level color coding so you can spot where tokenizers diverge -- Total token count + character compression ratio per tokenizer -- A "rare token" view that highlights tokens that decode to multiple Unicode codepoints -- Side-by-side cost projection across major API providers +Tokenizers decide how a model sees your text. The same prompt can produce very different token counts, and therefore cost and latency, across model families. This tool will run your text through several tokenizers in the browser and show the splits aligned column by column, with a token count and a cost projection per provider. ## Status -Experimental — currently runs entirely in-browser via WASM ports of the underlying libraries. Expect occasional Unicode edge cases. +Coming soon. Not yet available. Until then, the external tools on the playground page cover tokenization well. diff --git a/src/content/tools/training-compute-calc/TrainingComputeCalc.tsx b/src/content/tools/training-compute-calc/TrainingComputeCalc.tsx new file mode 100644 index 0000000..6037a1a --- /dev/null +++ b/src/content/tools/training-compute-calc/TrainingComputeCalc.tsx @@ -0,0 +1,315 @@ +'use client'; + +import { useState } from 'react'; +import { Field, Stat, StatRow } from '@/components/playground/primitives'; +import { + CHINCHILLA_TOKENS_PER_PARAM, + GPU_PEAK, + chinchillaTokensT, + formatCompact, + formatSci, + gpuHours, + impliedMfu, + tokensPerParam, + trainingFlops, + wallClockDays, + type GpuKey, +} from './compute'; + +type Preset = { + label: string; + params: number; + tokens: number; + gpu: GpuKey; + gpuExp?: number; + // GPU-hours the paper or model card reports, when one exists. + reportedGpuHours?: number; + reportedBy?: string; + mixedPrecision?: boolean; +}; + +const PRESETS: Preset[] = [ + { label: 'Chinchilla 70B', params: 70, tokens: 1.4, gpu: 'a100' }, + { + label: 'Llama 2 70B', + params: 70, + tokens: 2.0, + gpu: 'a100', + reportedGpuHours: 1_720_320, + reportedBy: 'Llama 2 model card', + }, + { label: 'Llama 3 8B', params: 8, tokens: 15, gpu: 'h100' }, + { + label: 'DeepSeek-V3 (37B active)', + params: 37, + tokens: 14.8, + gpu: 'h100', + gpuExp: 11, + reportedGpuHours: 2_664_000, + reportedBy: 'DeepSeek-V3 report, FP8 pretraining on H800s', + mixedPrecision: true, + }, + { + label: 'Llama 3 405B', + params: 405, + tokens: 15.6, + gpu: 'h100', + gpuExp: 14, + reportedGpuHours: 30_840_000, + reportedBy: 'Llama 3.1 model card, H100 training total', + }, +]; + +export default function TrainingComputeCalc({ compact = false }: { compact?: boolean }) { + const [params, setParams] = useState(70); + const [tokens, setTokens] = useState(1.4); + const [gpu, setGpu] = useState('h100'); + const [gpuExp, setGpuExp] = useState(10); + const [mfu, setMfu] = useState(40); + const [rate, setRate] = useState(2); + // The preset stays selected until another preset is chosen; sliders never clear it. + const [presetLabel, setPresetLabel] = useState(null); + + const gpuCount = 2 ** gpuExp; + const activePreset = PRESETS.find((p) => p.label === presetLabel); + // The paper's GPU-hours only compare cleanly while the model still matches the paper. + const presetIntact = + activePreset !== undefined && activePreset.params === params && activePreset.tokens === tokens; + const spec = GPU_PEAK[gpu]; + const flops = trainingFlops(params, tokens); + const hours = gpuHours(flops, spec.tflops, mfu / 100); + const days = wallClockDays(hours, gpuCount); + const ratio = tokensPerParam(params, tokens); + const sci = formatSci(flops); + const cost = hours * rate; + const effectiveTflops = spec.tflops * (mfu / 100); + const clusterPflops = (effectiveTflops * gpuCount) / 1000; + + const ratioNote = + ratio < CHINCHILLA_TOKENS_PER_PARAM * 0.75 + ? 'below the dense-model 20× reference' + : ratio > CHINCHILLA_TOKENS_PER_PARAM * 1.5 + ? 'above the dense-model 20× reference' + : 'near the dense-model 20× reference'; + + return ( +
+ {!compact && ( + <> +
+

+ Training Compute Calculator +

+ + · LIVE + +
+

+ How many FLOPs, GPU-hours, and days a pretraining run needs, from the 6ND rule and the + hardware you point at it. +

+ + )} + +
+ {PRESETS.map((p) => ( + + ))} +
+ +
+ + setParams(+e.target.value)} + style={{ width: '100%' }} + /> + + + setTokens(+e.target.value)} + style={{ width: '100%' }} + /> + + + + setGpuExp(+e.target.value)} + style={{ width: '100%' }} + /> + + + setMfu(+e.target.value)} + style={{ width: '100%' }} + /> + + +
+ {(Object.keys(GPU_PEAK) as GpuKey[]).map((k) => ( + + ))} +
+
+ + setRate(+e.target.value)} + style={{ width: '100%' }} + /> + +
+ +
+
+ Estimate +
+
+ + + = 1 ? `${days.toFixed(1)} days` : `${(days * 24).toFixed(1)} hours`} + /> +
+
+ + + + + + {presetIntact && activePreset?.reportedGpuHours && ( + + )} +
+
+ Approximate model compute using 6ND and dense BF16 peak throughput. MFU must reflect your + workload and cluster size; the estimate does not check memory fit or predict scaling + efficiency. Attention compute is omitted. Published GPU-hours are reference totals, not + predictions at the selected MFU. + {presetIntact && activePreset?.mixedPrecision && ( + <> + {' '} + DeepSeek used FP8 mixed precision on H800s. The selected H100 is a compute proxy; this + BF16 estimate is not a reproduction of that run. + + )} +
+
+
+ ); +} diff --git a/src/content/tools/training-compute-calc/compute.test.ts b/src/content/tools/training-compute-calc/compute.test.ts new file mode 100644 index 0000000..2a79bea --- /dev/null +++ b/src/content/tools/training-compute-calc/compute.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest'; +import { + chinchillaTokensT, + formatSci, + gpuHours, + impliedMfu, + tokensPerParam, + trainingFlops, + wallClockDays, +} from './compute'; + +describe('training compute math', () => { + it('reproduces the Llama 3 405B budget from the paper (3.8e25 FLOPs)', () => { + const flops = trainingFlops(405, 15.6); + expect(flops / 3.8e25).toBeCloseTo(1, 1); + }); + + it('converts 405B compute to hours at an assumed 34.6% utilization', () => { + const hours = gpuHours(trainingFlops(405, 15.6), 989, 0.346); + expect(hours / 30.84e6).toBeCloseTo(1, 1); + }); + + it('calculates the BF16 reference ratio from Llama 2 reported hours', () => { + const mfu = impliedMfu(trainingFlops(70, 2.0), 312, 1_720_320); + expect(mfu).toBeGreaterThan(0.4); + expect(mfu).toBeLessThan(0.47); + }); + + it('inverts GPU-hour accounting without treating the inferred ratio as a measurement', () => { + const flops = trainingFlops(37, 14.8); + const mfu = impliedMfu(flops, 989, 2_664_000); + expect(mfu).toBeGreaterThan(0.3); + expect(mfu).toBeLessThan(0.4); + const days = wallClockDays(gpuHours(flops, 989, mfu), 2048); + expect(days / (14.8 * 3.7)).toBeCloseTo(1, 1); + }); + + it('gives Chinchilla 70B its 1.4T tokens', () => { + expect(chinchillaTokensT(70)).toBeCloseTo(1.4, 5); + expect(tokensPerParam(70, 1.4)).toBeCloseTo(20, 5); + }); + + it('keeps the 20x token budget positive and exact across the parameter slider range', () => { + for (let params = 1; params <= 1000; params++) { + const tokens = chinchillaTokensT(params); + expect(tokens).toBeGreaterThanOrEqual(0.01); + expect(tokens).toBeLessThanOrEqual(30); + expect(tokensPerParam(params, tokens)).toBeCloseTo(20, 10); + } + expect(chinchillaTokensT(1)).toBe(0.02); + expect(chinchillaTokensT(3)).toBe(0.06); + }); + + it('converts a hand-calculated compute budget into GPU hours', () => { + // 1B parameters and 1T tokens = 6e21 FLOPs. + // 100 TFLOPS at 50% supplies 5e13 FLOPs/s: 120M seconds. + expect(trainingFlops(1, 1)).toBe(6e21); + expect(gpuHours(6e21, 100, 0.5)).toBeCloseTo(33333.333333, 5); + }); + + it('halves time with double utilization or double GPUs, holding other inputs fixed', () => { + const work = trainingFlops(70, 1.4); + const hours = gpuHours(work, 989, 0.4); + expect(gpuHours(work, 989, 0.2)).toBeCloseTo(hours * 2); + expect(wallClockDays(hours, 2048)).toBe(wallClockDays(hours, 1024) / 2); + }); + + it('distinguishes the full reported 405B hours from the 54-day snapshot', () => { + expect(wallClockDays(30_840_000, 16_384)).toBeCloseTo(78.4302, 4); + }); + + it('spreads GPU-hours across the cluster', () => { + expect(wallClockDays(24_000, 1000)).toBe(1); + }); + + it('formats scientific notation', () => { + expect(formatSci(3.79e25)).toEqual({ mantissa: '3.79', exponent: 25 }); + expect(formatSci(0)).toEqual({ mantissa: '0', exponent: 0 }); + }); +}); diff --git a/src/content/tools/training-compute-calc/compute.ts b/src/content/tools/training-compute-calc/compute.ts new file mode 100644 index 0000000..d77e945 --- /dev/null +++ b/src/content/tools/training-compute-calc/compute.ts @@ -0,0 +1,52 @@ +// Peak dense BF16 tensor-core throughput from NVIDIA datasheets (no 2:4 sparsity). +export const GPU_PEAK = { + a100: { name: 'A100 80GB', tflops: 312 }, + h100: { name: 'H100 SXM', tflops: 989 }, + h200: { name: 'H200 SXM', tflops: 989 }, + b200: { name: 'B200', tflops: 2250 }, +} as const; +export type GpuKey = keyof typeof GPU_PEAK; + +// Hoffmann et al. (Chinchilla, 2022): compute-optimal training uses ~20 tokens per parameter. +export const CHINCHILLA_TOKENS_PER_PARAM = 20; + +// Kaplan et al. (2020): forward + backward ≈ 6 FLOPs per parameter per token. +export function trainingFlops(paramsB: number, tokensT: number): number { + return 6 * paramsB * 1e9 * tokensT * 1e12; +} + +export function chinchillaTokensT(paramsB: number): number { + return (paramsB * 1e9 * CHINCHILLA_TOKENS_PER_PARAM) / 1e12; +} + +export function tokensPerParam(paramsB: number, tokensT: number): number { + return (tokensT * 1e12) / (paramsB * 1e9); +} + +export function gpuHours(flops: number, peakTflops: number, mfu: number): number { + return flops / (peakTflops * 1e12 * mfu) / 3600; +} + +export function wallClockDays(totalGpuHours: number, gpuCount: number): number { + return totalGpuHours / gpuCount / 24; +} + +// MFU as defined in the PaLM paper: achieved model FLOPs over peak hardware FLOPs. +export function impliedMfu(flops: number, peakTflops: number, reportedGpuHours: number): number { + return flops / (peakTflops * 1e12 * reportedGpuHours * 3600); +} + +export function formatSci(n: number): { mantissa: string; exponent: number } { + if (!isFinite(n) || n <= 0) return { mantissa: '0', exponent: 0 }; + const exponent = Math.floor(Math.log10(n)); + const mantissa = n / 10 ** exponent; + return { mantissa: mantissa.toFixed(2), exponent }; +} + +export function formatCompact(n: number): string { + if (!isFinite(n)) return '—'; + if (n >= 1e9) return `${(n / 1e9).toFixed(2)}B`; + if (n >= 1e6) return `${(n / 1e6).toFixed(2)}M`; + if (n >= 1e3) return `${(n / 1e3).toFixed(1)}K`; + return n.toFixed(0); +} diff --git a/src/content/tools/training-compute-calc/index.mdx b/src/content/tools/training-compute-calc/index.mdx new file mode 100644 index 0000000..0482e96 --- /dev/null +++ b/src/content/tools/training-compute-calc/index.mdx @@ -0,0 +1,55 @@ +--- +name: Training Compute Calculator +summary: Turn a model size and token budget into training FLOPs, GPU-hours, wall-clock days, and cost, using the 6ND rule and real GPU peak numbers. +tag: Live +icon: ∑ +authors: + - dinesh +topics: + - training + - distributed +tags: [training, flops, scaling-laws, chinchilla, mfu] +core: true +featured: true +--- + +import TrainingComputeCalc from './TrainingComputeCalc'; + + + +## What it does + +Pick a parameter count, a token budget, a GPU and a cluster size. The tool returns the total training compute, how many GPU-hours that is at a given utilization, how long the run takes on your cluster, and what it costs at your hourly rate. + +## The math + +- **Training FLOPs ≈ 6 × N × D**, where N is the parameter count and D is the training token count. The calculator converts billions of parameters and trillions of tokens to raw counts. This approximates the forward and backward parameter-matrix operations; it is not a complete operation count. +- **GPU-hours = FLOPs ÷ (peak FLOPS × MFU) ÷ 3600.** TFLOPS are multiplied by 10¹². MFU is entered as a percentage and converted to a fraction. +- **Days = GPU-hours ÷ GPU count ÷ 24.** This assumes the chosen MFU holds at the chosen cluster size. More GPUs do not automatically preserve utilization. +- **Cost = GPU-hours × hourly rate.** This is GPU rental cost only. + +The **20 tokens per parameter** button is a rough dense-model reference inspired by [Chinchilla](https://arxiv.org/abs/2203.15556), not a universal optimum. Data, architecture, and the intended inference budget change the best allocation. Applying that ratio to active MoE parameters does not establish compute optimality. + +[PaLM's MFU accounting](https://arxiv.org/abs/2204.02311) includes an additional attention term: approximately 12 × layers × query heads × head dimension × sequence length per token. This calculator omits it. The error depends on model size and context length; even at 8K context it can be substantial for smaller models. For example, 32 layers, 32 heads, head dimension 128, and 8,192 tokens add about 12.9 billion FLOPs per token, or 27% on top of 6N for an 8B model under that convention. + +## Published reference runs + +| Run | Parameters × tokens | Approximate 6ND FLOPs | Reported GPU-hours | +| -------------- | ------------------- | --------------------- | ----------------------------- | +| Llama 2 70B | 70B × 2T | 8.4 × 10²³ | 1,720,320 A100 | +| Llama 3.1 405B | 405B × 15.6T | 3.79 × 10²⁵ | 30.84M H100 | +| DeepSeek-V3 | 37B active × 14.8T | 3.29 × 10²⁴ | 2.664M H800, pretraining only | + +Sources: [Llama 2 model card](https://huggingface.co/meta-llama/Llama-2-70b), [Llama 3.1 model card](https://huggingface.co/meta-llama/Llama-3.1-405B), [Llama 3 report](https://arxiv.org/abs/2407.21783), and [DeepSeek-V3 report](https://arxiv.org/abs/2412.19437). + +Dividing approximate 6ND compute by reported GPU-hours and BF16 peak gives about 43.5% for Llama 2 70B and 34.5% for Llama 3.1 405B. These are inferred averages, not independent validation of the estimator or directly measured in-run MFU. Reported totals can cover different training stages and operating conditions. + +For scale, 30.84M GPU-hours divided by a constant 16,384 GPUs is 78.4 days. Meta's 54-day reliability observation is a snapshot, not the full training duration. DeepSeek's 2.664M pretraining GPU-hours divided by 2,048 GPUs is 54.2 days. Selecting a preset keeps your chosen MFU and hourly rate, so the estimate need not equal the reported total. + +## Hardware and limitations + +- Peak values use dense BF16 throughput, without structured sparsity: A100 312 TFLOPS, H100 SXM and H200 SXM approximately 989, and B200 2,250. See [NVIDIA A100](https://www.nvidia.com/en-us/data-center/a100/), [H100](https://www.nvidia.com/en-us/data-center/h100/), [H200](https://www.nvidia.com/en-us/data-center/h200/), and [NVIDIA's non-sparse throughput table](https://github.com/NVIDIA/exemplar-performance#gpu-peak-theoretical-throughput). +- The DeepSeek preset uses an H100 compute proxy for an H800 FP8 mixed-precision run. A BF16 denominator produces a _higher_, not lower, utilization percentage than a larger FP8 denominator for the same work and time. Mixed-precision utilization needs an explicit accounting convention, so this preset shows reported hours without an implied MFU comparison. +- For MoE, active parameters give only a rough compute estimate. Routing, attention, auxiliary objectives, and communication add work. Memory capacity still depends on total parameters and training state. +- Memory fit, batch size, parallelism, interconnect, failures, checkpointing, and data loading are not modeled independently. Choose an end-to-end MFU appropriate to those conditions; the calculator cannot determine whether the configuration is feasible. +- Rental cost excludes storage, networking, and separately billed failed or experimental runs. diff --git a/src/pages/index.astro b/src/pages/index.astro index 9241ff5..4fbf05d 100644 --- a/src/pages/index.astro +++ b/src/pages/index.astro @@ -16,22 +16,14 @@ const reading = allPosts.filter((post) => post.id !== lead?.id).slice(0, 3); const counts = countPostsByTopic(allPosts); const topics = TOPICS.filter((topic) => counts[topic.id] > 0).slice(0, 3); const moreTopics = TOPICS.filter((topic) => !topics.some((featured) => featured.id === topic.id)); -const liveTools = rankTools( - await getCollection('tools', ({ data }) => !data.draft && data.tag === 'Live'), -); -const tools = pickCoreTools(liveTools, 2); -const moreTools = liveTools.filter((tool) => !tools.some((featured) => featured.id === tool.id)); -const toolDescriptions: Record = { - 'attention-viz': - 'Explore a token-by-token attention map, from causal masking to recurring attention patterns.', - 'throughput-calc': - 'Estimate memory-bound token throughput and KV-cache size across models, precisions, batch sizes, and GPUs.', -}; +const allTools = rankTools(await getCollection('tools', ({ data }) => !data.draft)); +const tools = pickCoreTools(allTools, 4); +const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.id === tool.id)); ---
@@ -102,12 +94,7 @@ const toolDescriptions: Record = { {post.data.readMin} min read

- - {post.data.title} - - + {post.data.title}

{topic.desc}

- - {counts[topic.id]} {counts[topic.id] === 1 ? 'article' : 'articles'} - - + {counts[topic.id]} {counts[topic.id] === 1 ? 'article' : 'articles'} ))} @@ -193,7 +177,7 @@ const toolDescriptions: Record = {

{tool.data.name}

-

{toolDescriptions[tool.id] ?? tool.data.summary}

+

{tool.data.summary}

))} @@ -203,7 +187,12 @@ const toolDescriptions: Record = {

More to try

@@ -222,11 +211,13 @@ const toolDescriptions: Record = {

A useful explanation. A hard-won lesson. A question worth exploring. If you're learning or building ML systems, there's a place for your perspective here. -

Write with us Meet the community +

@@ -468,17 +459,6 @@ const toolDescriptions: Record = { line-height: 1.3; margin: 10px 0 12px; } - .reading-item h3 a { - display: flex; - justify-content: space-between; - align-items: baseline; - gap: 20px; - } - .reading-arrow { - font-family: var(--font-sans); - color: var(--ink-3); - font-size: 18px; - } .topic-section:has(.publication-more-topics), .tools-section:has(.publication-more-tools) { padding-bottom: calc(var(--section-space) - 22px); @@ -530,14 +510,9 @@ const toolDescriptions: Record = { line-height: 1.6; } .topic-end { - display: flex; - gap: 30px; - align-items: center; font-size: 12px; color: var(--ink-3); - } - .topic-end > :last-child { - font-size: 20px; + white-space: nowrap; } .publication-tools { display: grid; @@ -573,11 +548,29 @@ const toolDescriptions: Record = { } .publication-invitation { border-top: 1px solid var(--line-2); - padding: var(--section-space) 0; + position: relative; + isolation: isolate; + padding: clamp(64px, 7vw, 96px) 24px; display: grid; - grid-template-columns: 1.15fr 1fr; - gap: 64px; - align-items: center; + justify-items: center; + gap: 24px; + text-align: center; + } + .publication-invitation::before { + content: ''; + position: absolute; + inset: 0; + z-index: -1; + pointer-events: none; + background: radial-gradient( + ellipse at 50% 48%, + color-mix(in srgb, var(--accent) 9%, transparent), + color-mix(in srgb, var(--accent) 3%, transparent) 45%, + transparent 75% + ); + } + .publication-invitation .publication-label { + justify-content: center; } .publication-invitation h2 { font-family: var(--font-display); @@ -589,23 +582,29 @@ const toolDescriptions: Record = { } .invitation-copy p { color: var(--ink-2); - font-size: 18px; + font-size: 16px; line-height: 1.7; - max-width: 420px; - margin: 0 0 22px; + max-width: 540px; + margin: 0 auto 24px; + } + .invitation-actions { + display: flex; + align-items: center; + justify-content: center; + flex-wrap: wrap; + gap: 12px 28px; } .invitation-copy .publication-small-link { display: flex; width: fit-content; - margin-top: 12px; + margin-top: 0; } @media (max-width: 1000px) { .publication-hero { gap: 24px; } .reading-layout, - .publication-tools, - .publication-invitation { + .publication-tools { gap: 36px; } .lead-story { @@ -645,8 +644,7 @@ const toolDescriptions: Record = { gap: 6px; } .reading-layout, - .publication-tools, - .publication-invitation { + .publication-tools { grid-template-columns: 1fr; gap: 36px; } @@ -676,12 +674,6 @@ const toolDescriptions: Record = { .publication-topic h3 { font-size: 24px; } - .topic-end { - gap: 12px; - } - .topic-end > :first-child { - display: none; - } .publication-tool { gap: 20px; } diff --git a/src/styles/publication.css b/src/styles/publication.css index 24deafd..09c4548 100644 --- a/src/styles/publication.css +++ b/src/styles/publication.css @@ -597,32 +597,38 @@ input[type='radio'] { } .publication-topic-chips { display: flex; - gap: 10px; - overflow-x: auto; - padding: 3px 3px 6px; - scrollbar-width: thin; - scrollbar-color: var(--line-2) transparent; + flex-wrap: wrap; + align-items: baseline; + font-family: var(--font-read); + font-size: 17px; + letter-spacing: -0.01em; + color: var(--ink-2); } .publication-topic-chips a { - flex: 0 0 auto; - padding: 9px 16px; - border: 1px solid var(--line); - border-radius: 999px; - color: var(--ink-2); - font-size: 14px; - line-height: 1.4; + color: inherit; + padding: 2px 0; text-decoration: none; - transition: - border-color 150ms, - color 150ms; + transition: color 150ms; +} +.publication-topic-chips a:not(:last-child)::after { + content: '·'; + margin: 0 14px; + color: var(--ink-4); } .publication-topic-chips a:hover { - border-color: var(--accent); color: var(--accent); } .publication-topic-chips a:focus-visible { outline: 2px solid var(--accent); - outline-offset: 1px; + outline-offset: 3px; +} +.publication-chip-tag { + margin-left: 8px; + font: 10px/1 var(--font-mono); + letter-spacing: 0.08em; + text-transform: uppercase; + color: var(--ink-3); + vertical-align: middle; } .nav-topics { From fc8be4a6bae3a1e5b971bab56d46834916b17a1e Mon Sep 17 00:00:00 2001 From: Dinesh <13635627+HumbleBee14@users.noreply.github.com> Date: Wed, 16 Sep 2026 12:21:47 -0700 Subject: [PATCH 2/4] Tools layout fix --- src/pages/index.astro | 108 +++++++++++++++++++++++++++++++++---- src/styles/publication.css | 34 ++++++++++-- 2 files changed, 130 insertions(+), 12 deletions(-) diff --git a/src/pages/index.astro b/src/pages/index.astro index 4fbf05d..d8106cb 100644 --- a/src/pages/index.astro +++ b/src/pages/index.astro @@ -162,7 +162,7 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i tools.length > 0 && (
-

ML systems, hands-on.

+

ML Systems tools

All tools @@ -459,8 +459,7 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i line-height: 1.3; margin: 10px 0 12px; } - .topic-section:has(.publication-more-topics), - .tools-section:has(.publication-more-tools) { + .topic-section:has(.publication-more-topics) { padding-bottom: calc(var(--section-space) - 22px); } .topic-section .publication-more-topics, @@ -516,13 +515,79 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i } .publication-tools { display: grid; - grid-template-columns: 1fr 1fr; - gap: 64px; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 0 48px; + } + .tools-section { + position: relative; + isolation: isolate; + margin-top: 32px; + padding: clamp(40px, 4vw, 56px) 0 32px; + } + .tools-section::before { + content: ''; + position: absolute; + inset: 0 -24px; + z-index: -1; + pointer-events: none; + background: color-mix(in srgb, var(--accent) 3%, var(--paper)); + } + .tools-section .publication-section-head { + display: grid; + grid-template-columns: 1fr auto 1fr; + align-items: center; + margin-bottom: 40px; + } + .tools-section .publication-section-head h2 { + grid-column: 2; + text-align: center; + max-width: none; + } + .tools-section .publication-section-head > a { + justify-self: end; } .publication-tool { + position: relative; + isolation: isolate; display: flex; align-items: flex-start; gap: 24px; + padding: 24px 0; + } + .publication-tool::before { + content: ''; + position: absolute; + inset: 8px -12px; + z-index: -1; + border: 1px solid color-mix(in srgb, var(--accent) 35%, var(--line)); + border-radius: 10px; + opacity: 0; + pointer-events: none; + transition: opacity 160ms ease; + } + .publication-tool:hover::before, + .publication-tool:focus-visible::before { + opacity: 1; + } + .publication-tool:nth-child(-n + 2)::before { + top: -12px; + } + .publication-tool:nth-last-child(-n + 2)::before { + bottom: -12px; + } + @media (prefers-reduced-motion: reduce) { + .publication-tool::before { + transition: none; + } + } + .publication-tool:nth-child(-n + 2) { + padding-top: 0; + } + .publication-tool:nth-child(n + 3) { + border-top: 1px solid var(--line); + } + .publication-tool:nth-last-child(-n + 2) { + padding-bottom: 0; } .publication-tool-icon { flex: 0 0 56px; @@ -603,8 +668,7 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i .publication-hero { gap: 24px; } - .reading-layout, - .publication-tools { + .reading-layout { gap: 36px; } .lead-story { @@ -617,6 +681,17 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i } } @media (max-width: 720px) { + .tools-section .publication-section-head { + grid-template-columns: 1fr; + gap: 4px; + margin-bottom: 28px; + } + .tools-section .publication-section-head h2 { + grid-column: 1; + } + .tools-section .publication-section-head > a { + justify-self: center; + } .publication-hero { grid-template-columns: 1fr; padding: 28px 0 var(--section-space); @@ -659,8 +734,7 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i .reading-item h3 { font-size: 23px; } - .topic-section:has(.publication-more-topics), - .tools-section:has(.publication-more-tools) { + .topic-section:has(.publication-more-topics) { padding-bottom: calc(var(--section-space) - 16px); } .topic-section .publication-more-topics, @@ -677,6 +751,22 @@ const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.i .publication-tool { gap: 20px; } + .publication-tools { + gap: 0; + } + .publication-tool:nth-child(n + 2) { + padding-top: 24px; + border-top: 1px solid var(--line); + } + .publication-tool:not(:last-child) { + padding-bottom: 24px; + } + .publication-tool:nth-child(n + 2)::before { + top: 8px; + } + .publication-tool:not(:last-child)::before { + bottom: 8px; + } .publication-tool h3 { font-size: 19px; } diff --git a/src/styles/publication.css b/src/styles/publication.css index 09c4548..4b78d8e 100644 --- a/src/styles/publication.css +++ b/src/styles/publication.css @@ -605,22 +605,50 @@ input[type='radio'] { color: var(--ink-2); } .publication-topic-chips a { + position: relative; color: inherit; padding: 2px 0; text-decoration: none; transition: color 150ms; } +.publication-topic-chips a::before { + content: ''; + position: absolute; + inset: -4px -7px; + border: 1px solid color-mix(in srgb, var(--accent) 35%, var(--line)); + border-radius: 6px; + opacity: 0; + pointer-events: none; + transition: opacity 160ms ease; +} +.publication-topic-chips a:not(:last-child) { + margin-right: 32px; +} .publication-topic-chips a:not(:last-child)::after { content: '·'; - margin: 0 14px; + position: absolute; + right: -18px; color: var(--ink-4); } +.publication-topic-chips a:hover::before, +.publication-topic-chips a:focus-visible::before { + opacity: 1; +} .publication-topic-chips a:hover { color: var(--accent); } .publication-topic-chips a:focus-visible { - outline: 2px solid var(--accent); - outline-offset: 3px; + outline: none; + color: var(--accent); +} +.publication-topic-chips a:focus-visible::before { + border-color: var(--accent); +} +@media (prefers-reduced-motion: reduce) { + .publication-topic-chips a, + .publication-topic-chips a::before { + transition: none; + } } .publication-chip-tag { margin-left: 8px; From 1ca750290db320b7ac4b4b04b2338109c9adadf6 Mon Sep 17 00:00:00 2001 From: Dinesh <13635627+HumbleBee14@users.noreply.github.com> Date: Wed, 16 Sep 2026 12:25:11 -0700 Subject: [PATCH 3/4] Update about.astro --- src/pages/about.astro | 30 +++++++++++++----------------- 1 file changed, 13 insertions(+), 17 deletions(-) diff --git a/src/pages/about.astro b/src/pages/about.astro index 52e619d..44a5c26 100644 --- a/src/pages/about.astro +++ b/src/pages/about.astro @@ -26,7 +26,7 @@ const links = [
@@ -38,33 +38,29 @@ const links = [

- {SITE.name} is an open community of contributors — engineers, researchers, students, - and curious learners — writing down what we figure out about modern machine learning systems and - sharing it with the world. + {SITE.name} is an open community of engineers, researchers, students, and curious + learners. We share what we learn about machine learning systems so others can build on it.

- The goal is simple: collect honest, technically grounded writing about ML systems — the - kernels, the schedulers, the embeddings, the deployments — without the marketing gloss. - Anyone interested in this domain is welcome to write here, whether you've shipped systems - for years or you just figured something out last week. If you've learned something worth - passing on, come write with us. + You'll find practical explanations and lessons from building ML systems, from kernels and + schedulers to training and deployment. We value clear writing, technical depth, and honest + accounts of what worked and what didn't.

- We think this knowledge is about to matter much more than it does today — - here's why this exists. + Whether you've built systems for years or just figured something out, your experience can + help someone else. Everyone is welcome to contribute.

- You don't need an invitation to contribute. Open the editor, share an - explanation or experience, and submit it for review. Every article carries its author's - name; every useful perspective helps the community learn. + Open the editor, share an explanation or experience, and submit it for + review. Every article is credited to its author.

-

- Everything here is © its respective author, all rights reserved. Notes from the workbench. -

+

Why we're building this community.

+ +

Content belongs to its respective author. All rights reserved.

- Back-of-envelope tokens/sec for a given model, precision, and hardware. Memory-bound - regime only; assumes batched serving with a healthy KV cache headroom. + A weight-bandwidth ceiling, not measured throughput. KV memory uses a fixed example: 80 + layers, 64 KV heads, head dimension 128, and BF16 cache. Model size changes weights + only.

)} @@ -72,6 +73,7 @@ export default function ThroughputCalc({ compact = false }: { compact?: boolean > Estimate
-
- - - +
+ + +
-
- ⚠ Estimate is memory-bound roofline only. Actual numbers depend on kernel quality, - continuous batching, speculative decoding, and a dozen other things this tool doesn't - model. + ⚠ The ceiling ignores KV-cache traffic and compute limits. Precision describes weight + storage, not native GPU support. Actual numbers depend on kernel quality, continuous + batching, speculative decoding, and a dozen other things this tool doesn't model.
diff --git a/src/content/tools/throughput-calc/index.mdx b/src/content/tools/throughput-calc/index.mdx index fd4949d..26fdf4a 100644 --- a/src/content/tools/throughput-calc/index.mdx +++ b/src/content/tools/throughput-calc/index.mdx @@ -1,6 +1,6 @@ --- name: Throughput Calculator -summary: Estimate memory-bound token throughput and KV-cache size across models, precisions, batch sizes, and GPUs. +summary: Explore a weight-bandwidth ceiling and a fixed KV-cache memory example across precisions, batch sizes, and GPUs. tag: Live icon: ◐ authors: @@ -18,14 +18,15 @@ import ThroughputCalc from './ThroughputCalc'; ## What it does -Back-of-the-envelope throughput estimation that's accurate enough to inform real architecture decisions. Plug in a model, precision, batch size, and GPU. Get tokens/sec, KV-cache size, and a verdict on whether you'll fit on one card. +Shows an ideal weight-read ceiling: GPU memory bandwidth divided by weight bytes, multiplied by batch size. It is an upper bound for a simplified decode step, not a throughput prediction. -## Why it's useful - -Most "how fast will this run" questions can be answered without standing up infrastructure. This tool encodes the memory-bound roofline that governs LLM inference, so you can compare options at the cost of one keystroke instead of one cluster-hour. +The memory example assumes 80 layers, 64 KV heads, a head dimension of 128, and a BF16 cache. These dimensions stay fixed when you change the parameter count. Use the Training Memory Calculator for architecture-specific memory estimates. ## Limitations -- Memory-bound regime only. Doesn't model the compute-bound prefill ceiling. -- Doesn't account for continuous batching, speculative decoding, or prefix sharing — those are explicit knobs in real serving stacks. -- Use the result as a rule of thumb, not a procurement quote. +- The ceiling omits KV-cache traffic, compute limits, communication, and kernel overhead. Longer sequences increase the displayed memory but do not change this weight-only ceiling. +- Precision sets weight storage size. It does not establish that a GPU supports native arithmetic at that precision or that suitable kernels exist. +- Memory fit reserves 15% headroom. A configuration that needs sharding cannot achieve the displayed ceiling on a single GPU; multi-GPU performance is not modeled. +- Use measured benchmarks for deployment decisions. + +See [How To Scale Your Model: Transformer Inference](https://jax-ml.github.io/scaling-book/inference/) for the full bandwidth and compute model. diff --git a/src/pages/about.astro b/src/pages/about.astro index 44a5c26..75fd1f6 100644 --- a/src/pages/about.astro +++ b/src/pages/about.astro @@ -2,6 +2,7 @@ import BaseLayout from '@/layouts/BaseLayout.astro'; import PageIntro from '@/components/PageIntro.astro'; import FounderBee from '@/components/FounderBee.astro'; +import CommunityIllustration from '@/components/CommunityIllustration.astro'; import IconLinkCard from '@/components/IconLinkCard.astro'; import { SITE } from '@/lib/site'; @@ -36,49 +37,70 @@ const links = [

-
-

- {SITE.name} is an open community of engineers, researchers, students, and curious - learners. We share what we learn about machine learning systems so others can build on it. -

+
+
+

+ {SITE.name} is an open community of engineers, researchers, students, and curious + learners. We share what we learn about machine learning systems so others can build on it. +

-

- You'll find practical explanations and lessons from building ML systems, from kernels and - schedulers to training and deployment. We value clear writing, technical depth, and honest - accounts of what worked and what didn't. -

+

+ You'll find practical explanations and lessons from building ML systems, from kernels and + schedulers to training and deployment. We value clear writing, technical depth, and honest + accounts of what worked and what didn't. +

-

- Whether you've built systems for years or just figured something out, your experience can - help someone else. Everyone is welcome to contribute. -

+

+ Whether you've built systems for years or just figured something out, your experience can + help someone else. Everyone is welcome to contribute. +

-

- Open the editor, share an explanation or experience, and submit it for - review. Every article is credited to its author. -

+

+ Open the editor, share an explanation or experience, and submit it + for review. Every article is credited to its author. +

-

Why we're building this community.

+

Why we're building this community.

-

Content belongs to its respective author. All rights reserved.

+

Content belongs to its respective author. All rights reserved.

- + -

- Created and maintained by Humble Bee. - -

+

+ Created and maintained by Humble Bee. + +

+
+
diff --git a/src/pages/content-policy.astro b/src/pages/content-policy.astro index 2f8c958..3f5bd5b 100644 --- a/src/pages/content-policy.astro +++ b/src/pages/content-policy.astro @@ -1,5 +1,6 @@ --- import BaseLayout from '@/layouts/BaseLayout.astro'; +import PageIntro from '@/components/PageIntro.astro'; import { SITE } from '@/lib/site'; const canonical = `${SITE.url}/content-policy`; @@ -11,11 +12,8 @@ const canonical = `${SITE.url}/content-policy`; canonical={canonical} >
-
-
Policy
-

Content policy.

-

Last updated July 2026

-
+ +

Last updated July 2026

@@ -114,28 +112,17 @@ const canonical = `${SITE.url}/content-policy`; diff --git a/src/pages/playground/[tool].astro b/src/pages/playground/[tool].astro index 2b0f33f..0281c3d 100644 --- a/src/pages/playground/[tool].astro +++ b/src/pages/playground/[tool].astro @@ -106,6 +106,7 @@ const breadcrumbJsonLd = { } .tool-meta { display: flex; + flex-wrap: wrap; align-items: center; gap: 12px; margin-bottom: 24px; diff --git a/src/pages/playground/index.astro b/src/pages/playground/index.astro index a643062..321abf0 100644 --- a/src/pages/playground/index.astro +++ b/src/pages/playground/index.astro @@ -15,7 +15,7 @@ const coreTools = pickCoreTools(allTools).map((t) => ({ name: t.data.name, desc: t.data.summary, tag: t.data.tag, - available: t.data.tag === 'Live', + available: t.data.tag !== 'Soon', })); const catalogTools = rankTools(allTools); diff --git a/src/pages/privacy.astro b/src/pages/privacy.astro index 5456431..b2dbc99 100644 --- a/src/pages/privacy.astro +++ b/src/pages/privacy.astro @@ -1,5 +1,6 @@ --- import BaseLayout from '@/layouts/BaseLayout.astro'; +import PageIntro from '@/components/PageIntro.astro'; import { SITE } from '@/lib/site'; const canonical = `${SITE.url}/privacy`; @@ -11,11 +12,8 @@ const canonical = `${SITE.url}/privacy`; canonical={canonical} >

-
-
Policy
-

Privacy.

-

Last updated July 2026

-
+ +

Last updated September 2026

@@ -25,10 +23,11 @@ const canonical = `${SITE.url}/privacy`;

No accounts, no logins

- You can read everything without signing in. The in-browser writing tool at /write runs entirely in your browser — what you type stays on your device, and nothing is sent to us - until you choose to download it and submit it by pull request, issue, or email. + You can read everything without signing in. Drafts in the writing tool at /write + are saved on your device. When you submit an article for review, its content, attached files, + and author details are sent to our publishing service to create a public pull request on GitHub. + You can also download your draft and submit it yourself by pull request, issue, or email.

Analytics

@@ -82,27 +81,16 @@ const canonical = `${SITE.url}/privacy`; diff --git a/src/pages/tags/[tag]/[...page].astro b/src/pages/tags/[tag]/[...page].astro index 1097090..66a4515 100644 --- a/src/pages/tags/[tag]/[...page].astro +++ b/src/pages/tags/[tag]/[...page].astro @@ -3,6 +3,7 @@ import type { GetStaticPaths, Page } from 'astro'; import { getCollection } from 'astro:content'; import BaseLayout from '@/layouts/BaseLayout.astro'; import Pager from '@/components/Pager.astro'; +import PageIntro from '@/components/PageIntro.astro'; import PostRow from '@/components/PostRow.astro'; import { sortPostsByDate, tagSlug } from '@/lib/data'; import { resolvePostAuthors, type PostWithAuthors } from '@/lib/posts'; @@ -79,16 +80,12 @@ const title = jsonLd={[collectionJsonLd, breadcrumbJsonLd]} >
-
-
-
Tag
-

#{label}

-

- {page.total} - {page.total === 1 ? 'article' : 'articles'} tagged {label}. -

-
-
+ +

+ {page.total} + {page.total === 1 ? 'article' : 'articles'} tagged {label}. +

+
{page.data.map((a) => )} diff --git a/src/pages/why.astro b/src/pages/why.astro index bf51875..32927b5 100644 --- a/src/pages/why.astro +++ b/src/pages/why.astro @@ -10,7 +10,7 @@ const pageJsonLd = { '@type': 'WebPage', name: 'Why this exists', description: - 'Why learning machine learning systems matters: intelligence is moving to personal, local devices — and the systems layer decides how fast we get there.', + 'Why learning machine learning systems matters: intelligence is moving to personal, local devices, and systems engineering helps make that possible.', url: canonical, isPartOf: { '@type': 'WebSite', name: SITE.name, url: SITE.url }, }; @@ -18,76 +18,68 @@ const pageJsonLd = {
- + +

+ Understanding the systems behind AI helps more people build, question, and improve them. +

+

- Every important technology ends up boring. Electricity, databases, GPS — miracles that - became plumbing. Machine intelligence is on the same path, and we are living through its - plumbing years. This site is about those years, and the people doing the work. + ML systems turn model capabilities into something people can use. This site is for the + people learning how that happens and sharing what they discover.

-

Most "model progress" is systems progress

+

Systems make models useful

- Ask what actually changed between the demo that amazed you and the product you use every - day: tokens got cheaper, first tokens got faster, contexts got longer, models started - fitting on hardware you own. Almost none of that came from smarter weights. It came from - quantization, batching, caches, kernels, schedulers — the unglamorous layer underneath. - The distance between a demo and a product is measured in milliseconds and megabytes, and - systems engineers are the ones who close it. + A capable model is only part of a working product. Memory use, response time, training + cost, and reliability matter too. Quantization, batching, caches, kernels, and + schedulers help make models practical. We want to make that work easier to understand.

-

Computing always moves closer to you

+

More choice in where AI runs

- Mainframe to desktop, desktop to pocket, cloud to edge — every generation of computing - ends up nearer to the person using it, because latency, cost, and privacy all pull the - same way. Intelligence is on the same road. The endpoint is a model that runs on devices - you own, tuned on your own context — your notes, your work, your family's routines — a - private intelligence layer that answers to you and no one else. A model that knows you - that well shouldn't live in someone else's building. The datacenter era of AI is its - mainframe era, and the people who understand inference at the edge are the ones who will - end it. + Some workloads belong in a datacenter. Others benefit from running on a laptop, phone, + or device nearby. Local inference can offer privacy, offline access, and lower latency, + with real limits on memory and power. Understanding those tradeoffs gives people more + control over the systems they use.

-

The fundamentals outlast the headlines

+

Fundamentals outlast the headlines

- Architectures churn monthly; the systems layer barely moves. Memory hierarchies, - arithmetic intensity, batching tradeoffs, the cost of moving a byte versus computing on - it — these were true before transformers and will be true after them. Learning ML - systems is learning the invariants: knowledge that compounds for decades while the - leaderboards reshuffle. + Models and frameworks change quickly. Memory hierarchies, arithmetic intensity, + batching, and the cost of moving data remain useful ways to reason about them. Learning + these fundamentals helps you evaluate new ideas instead of starting from scratch each + time.

-

The bottleneck is people

+

Knowledge grows when we share it

- The knowledge that makes all of this work is concentrated in a handful of infrastructure - teams and scattered across conference talks and half-finished blog posts. That scarcity - is the real constraint on how fast the local, personal future arrives. The fix is old - and reliable: write things down, in the open, where anyone can learn them. A field grows - exactly as fast as its commons. + Useful knowledge is scattered across papers, code, talks, and individual experience. A + clear explanation or an honest account of a failed approach can save someone else days + of work. Publishing it openly makes that experience available beyond one team.

-

So we write

+

A place to contribute

- Articles, primers, and tools from practitioners — honest, technically grounded, free to - read, open to anyone who has figured something out and is willing to pass it on. If the - future we described sounds right to you, help build the commons that gets us there - sooner. + We bring together articles, primers, and tools from people learning and building ML + systems. Everything is free to read, and anyone can submit work for review. If you've + learned something worth passing on, there's room for it here.

@@ -194,14 +186,14 @@ const pageJsonLd = { } .why-lede { font-family: var(--font-read); - font-style: italic; + font-style: normal; font-size: 21px; line-height: 1.6; color: var(--ink); margin: 8px 0 0; } .why-body section { - margin-top: 48px; + margin-top: 36px; } .why-body h2 { font-family: var(--font-display); @@ -221,6 +213,6 @@ const pageJsonLd = { display: flex; gap: 12px; flex-wrap: wrap; - margin-top: 56px; + margin-top: 36px; } diff --git a/src/styles/publication.css b/src/styles/publication.css index 4b78d8e..83bb3f3 100644 --- a/src/styles/publication.css +++ b/src/styles/publication.css @@ -131,6 +131,9 @@ main { padding-top: 56px; } .footer-tagline { + font-family: var(--font-sans); + font-style: normal; + font-weight: 400; max-width: 320px; font-size: 15px; line-height: 1.7; @@ -900,3 +903,12 @@ input[type='radio'] { transition: none; } } + +/* Keep schematic frames legible against the dark hero without changing data colors. */ +[data-theme='dark'] .hero-scenes { + --line: color-mix(in srgb, var(--ink-3) 48%, var(--paper)); + --line-2: color-mix(in srgb, var(--ink-3) 68%, var(--paper)); +} +[data-theme='dark'] .hero-scenes line[stroke='var(--ink-3)'][stroke-width='0.5'] { + stroke-width: 0.85; +}