diff --git a/src/components/CommunityIllustration.astro b/src/components/CommunityIllustration.astro new file mode 100644 index 0000000..4fa3944 --- /dev/null +++ b/src/components/CommunityIllustration.astro @@ -0,0 +1,151 @@ +--- +const nodes = [ + { x: 98, y: 108, delay: 0.8 }, + { x: 314, y: 90, delay: 1.2 }, + { x: 350, y: 245, delay: 1.6 }, + { x: 270, y: 355, delay: 2 }, + { x: 88, y: 320, delay: 2.4 }, + { x: 48, y: 208, delay: 2.8 }, +]; +--- + + + + diff --git a/src/components/HeroFigure.tsx b/src/components/HeroFigure.tsx index 21ae5fc..122a24f 100644 --- a/src/components/HeroFigure.tsx +++ b/src/components/HeroFigure.tsx @@ -62,7 +62,7 @@ export function AttentionFig({ t }: { t: number }) { // Build the causal triangle in under two seconds, then scan the completed rows. const reveal = t / 0.14; const scan = t < 1.96 ? reveal : ((t - 1.96) % 2.8) / 0.2; - const activeRow = Math.min(N - 1, Math.floor(scan)); + const activeRow = Math.max(0, Math.min(N - 1, Math.floor(scan))); return ( ; + return ; } export default function HeroFigure() { diff --git a/src/components/Playground.tsx b/src/components/Playground.tsx index b960edc..a59627c 100644 --- a/src/components/Playground.tsx +++ b/src/components/Playground.tsx @@ -4,6 +4,7 @@ import { useState } from 'react'; import ThroughputCalc from '@/content/tools/throughput-calc/ThroughputCalc'; import GpuMemoryCalc from '@/content/tools/gpu-mem-calc/GpuMemoryCalc'; import AttentionViz from '@/content/tools/attention-viz/AttentionViz'; +import TrainingComputeCalc from '@/content/tools/training-compute-calc/TrainingComputeCalc'; export type PlaygroundTool = { id: string; @@ -17,6 +18,7 @@ const TOOL_COMPONENTS: Record = { 'throughput-calc': ThroughputCalc, 'gpu-mem-calc': GpuMemoryCalc, 'attention-viz': AttentionViz, + 'training-compute-calc': TrainingComputeCalc, }; export default function Playground({ tools }: { tools: PlaygroundTool[] }) { diff --git a/src/components/ToolGlyph.tsx b/src/components/ToolGlyph.tsx index 4318051..8bb0fb6 100644 --- a/src/components/ToolGlyph.tsx +++ b/src/components/ToolGlyph.tsx @@ -47,6 +47,22 @@ export default function ToolGlyph({ id }: { id: string }) { ); + case 'training-compute-calc': + return ( + + + + + + + ); case 'model-card': return ( diff --git a/src/content/tools/attention-viz/AttentionViz.tsx b/src/content/tools/attention-viz/AttentionViz.tsx index da3abea..9595d64 100644 --- a/src/content/tools/attention-viz/AttentionViz.tsx +++ b/src/content/tools/attention-viz/AttentionViz.tsx @@ -1,128 +1,58 @@ 'use client'; import { useEffect, useState } from 'react'; -import { Field } from '@/components/playground/primitives'; import { AttentionFig } from '@/components/HeroFigure'; -function useAnimationFrame() { - const [t, setT] = useState(0); +export default function AttentionViz() { + const [t, setT] = useState(5); + const [paused, setPaused] = useState(false); + const [reducedMotion, setReducedMotion] = useState(true); useEffect(() => { - let raf: number; - const start = performance.now(); - const tick = (now: number) => { - setT((now - start) / 1000); - raf = requestAnimationFrame(tick); + const query = window.matchMedia('(prefers-reduced-motion: reduce)'); + let frame = 0; + const update = () => { + cancelAnimationFrame(frame); + setReducedMotion(query.matches); + if (query.matches || paused) return; + const start = performance.now(); + const tick = (now: number) => { + setT((now - start) / 1000); + frame = requestAnimationFrame(tick); + }; + frame = requestAnimationFrame(tick); }; - raf = requestAnimationFrame(tick); - return () => cancelAnimationFrame(raf); - }, []); - return t; -} - -export default function AttentionViz({ compact = false }: { compact?: boolean }) { - const [model, setModel] = useState('llama-7b'); - const [layer, setLayer] = useState(14); - const [head, setHead] = useState(3); - const [prompt, setPrompt] = useState('The cat sat on the mat and the dog ran past him.'); - const t = useAnimationFrame(); + update(); + query.addEventListener('change', update); + return () => { + cancelAnimationFrame(frame); + query.removeEventListener('change', update); + }; + }, [paused]); return (
- {!compact && ( - <> -
-

- Attention Visualizer -

- - · LIVE - -
-

- Inspect attention patterns for any model. Drag prompts in, scrub layers and heads. -

- +

+ An illustrative attention map for a fixed sentence. These patterns are synthetic; no model + is loaded and no prompt is sent to a server. +

+ {!reducedMotion && ( + )} - -
- {['llama-7b', 'llama-70b', 'mistral-7b', 'qwen-72b'].map((m) => ( - - ))} -
-
-
- - setLayer(+e.target.value)} - style={{ width: '100%' }} - /> - - - setHead(+e.target.value)} - style={{ width: '100%' }} - /> - -
-
- - setPrompt(e.target.value)} - style={{ - width: '100%', - padding: '10px 12px', - background: 'var(--paper)', - border: '1px solid var(--line-2)', - borderRadius: 6, - fontFamily: 'var(--font-mono)', - fontSize: 13, - color: 'var(--ink)', - }} - /> - -
- +
); diff --git a/src/content/tools/attention-viz/index.mdx b/src/content/tools/attention-viz/index.mdx index c95d282..2a78243 100644 --- a/src/content/tools/attention-viz/index.mdx +++ b/src/content/tools/attention-viz/index.mdx @@ -1,7 +1,7 @@ --- name: Attention Visualizer -summary: Inspect attention patterns layer-by-layer for any Hugging Face model. Click any head to see its causal mask, induction behavior, and sink tokens. -tag: Live +summary: Explore a synthetic attention map illustrating causal masking and recurring patterns. +tag: Beta icon: ◎ authors: - lchen @@ -18,23 +18,10 @@ import AttentionViz from './AttentionViz'; -## What it does +## What it shows -Loads any model from Hugging Face and renders the attention weights produced for a prompt you control. Built for the moment when you're trying to convince yourself that a specific head is doing what you think it's doing. - -## Why it's useful - -Reading the attention matrix as raw numbers is hopeless. Reading it as a heatmap with the tokens labeled along both axes makes patterns jump out — causal triangles, sink tokens absorbing residual mass, induction heads aligning along the off-diagonal. A few minutes here often saves hours of grepping through paper figures. - -## How to use it - -1. Paste a model ID from Hugging Face (e.g. `meta-llama/Llama-3.1-8B`). -2. Type a prompt — short ones make the visualization legible. -3. Pick a layer and head from the sidebar. -4. Watch the matrix render. Hover any cell to see the query/key tokens. +A synthetic, animated attention matrix for a fixed sentence. Rows represent query positions and columns represent key positions. The triangular shape illustrates causal masking: a token cannot attend to future tokens. ## Limitations -- Currently runs on a backend GPU, so very large models may queue. -- Multi-query and grouped-query attention are visualized per query head; KV-shared heads appear repeated. -- This tool is for understanding, not for high-throughput batch analysis. +This is a visual demonstration, not measured attention from a trained model. Model loading, custom prompts, layer selection, and head inspection are not implemented. The colors should not be interpreted as evidence about any specific model. diff --git a/src/content/tools/eval-harness/index.mdx b/src/content/tools/eval-harness/index.mdx index c930fc0..49ad003 100644 --- a/src/content/tools/eval-harness/index.mdx +++ b/src/content/tools/eval-harness/index.mdx @@ -1,33 +1,20 @@ --- name: Eval Harness Playground -summary: Run a small set of evaluations against any inference endpoint and get back a structured scorecard — quality, latency, cost, and refusal rate side by side. -tag: Beta +summary: Run a small, fixed set of evals against an OpenAI-compatible endpoint and get a scorecard for quality, latency, and cost. +tag: Soon icon: ▤ authors: - - lchen + - dinesh topics: - evaluation tags: [evals, benchmarks, scorecards] core: true --- -## What it does +## What we are building -Points at any inference endpoint (yours or a hosted one), runs a curated set of small but informative evals, and produces a scorecard. The set is small on purpose: you get answers in minutes instead of hours, and the evals are chosen for signal-to-noise rather than headline numbers. +Full benchmark suites such as lm-evaluation-harness take hours and real money to run. Most of the time the question is simpler: is this endpoint worth a closer look? This tool will point at any OpenAI-compatible endpoint, run a small curated set of prompts, and return a scorecard with quality, latency, and cost side by side. -## Why it's useful +## Status -Full benchmark suites take hours and cost real money. Most of the time you just want a quick "is this model worth a closer look." This playground gives you a focused first read so you can decide whether to invest in a full eval pass. - -## How to use it - -1. Provide an endpoint URL and auth header (or pick a hosted preset). -2. Select an eval pack — reasoning, code, instruction following, refusal behavior. -3. Run. Results stream in as each prompt completes. -4. Export the scorecard or share a permalink. - -## Limitations - -- Beta: the eval set is opinionated and will keep evolving. -- Not a replacement for full evaluation suites when stakes are high. -- Caches results by prompt hash, so re-running the same prompts is free. +Coming soon. Not yet available. diff --git a/src/content/tools/gpu-mem-calc/GpuMemoryCalc.tsx b/src/content/tools/gpu-mem-calc/GpuMemoryCalc.tsx index e9023dd..59a4644 100644 --- a/src/content/tools/gpu-mem-calc/GpuMemoryCalc.tsx +++ b/src/content/tools/gpu-mem-calc/GpuMemoryCalc.tsx @@ -234,6 +234,7 @@ export default function GpuMemoryCalc({ compact = false }: { compact?: boolean } > @@ -56,8 +56,9 @@ export default function ThroughputCalc({ compact = false }: { compact?: boolean

- Back-of-envelope tokens/sec for a given model, precision, and hardware. Memory-bound - regime only; assumes batched serving with a healthy KV cache headroom. + A weight-bandwidth ceiling, not measured throughput. KV memory uses a fixed example: 80 + layers, 64 KV heads, head dimension 128, and BF16 cache. Model size changes weights + only.

)} @@ -72,6 +73,7 @@ export default function ThroughputCalc({ compact = false }: { compact?: boolean > Estimate -
- - - +
+ + +
-
- ⚠ Estimate is memory-bound roofline only. Actual numbers depend on kernel quality, - continuous batching, speculative decoding, and a dozen other things this tool doesn't - model. + ⚠ The ceiling ignores KV-cache traffic and compute limits. Precision describes weight + storage, not native GPU support. Actual numbers depend on kernel quality, continuous + batching, speculative decoding, and a dozen other things this tool doesn't model.
diff --git a/src/content/tools/throughput-calc/index.mdx b/src/content/tools/throughput-calc/index.mdx index c8f25b4..26fdf4a 100644 --- a/src/content/tools/throughput-calc/index.mdx +++ b/src/content/tools/throughput-calc/index.mdx @@ -1,6 +1,6 @@ --- name: Throughput Calculator -summary: Estimate tokens/sec for any model, precision, batch size, and GPU combination — memory-bound roofline only, no kernel-quality wishful thinking. +summary: Explore a weight-bandwidth ceiling and a fixed KV-cache memory example across precisions, batch sizes, and GPUs. tag: Live icon: ◐ authors: @@ -18,14 +18,15 @@ import ThroughputCalc from './ThroughputCalc'; ## What it does -Back-of-the-envelope throughput estimation that's accurate enough to inform real architecture decisions. Plug in a model, precision, batch size, and GPU. Get tokens/sec, KV-cache size, and a verdict on whether you'll fit on one card. +Shows an ideal weight-read ceiling: GPU memory bandwidth divided by weight bytes, multiplied by batch size. It is an upper bound for a simplified decode step, not a throughput prediction. -## Why it's useful - -Most "how fast will this run" questions can be answered without standing up infrastructure. This tool encodes the memory-bound roofline that governs LLM inference, so you can compare options at the cost of one keystroke instead of one cluster-hour. +The memory example assumes 80 layers, 64 KV heads, a head dimension of 128, and a BF16 cache. These dimensions stay fixed when you change the parameter count. Use the Training Memory Calculator for architecture-specific memory estimates. ## Limitations -- Memory-bound regime only. Doesn't model the compute-bound prefill ceiling. -- Doesn't account for continuous batching, speculative decoding, or prefix sharing — those are explicit knobs in real serving stacks. -- Use the result as a rule of thumb, not a procurement quote. +- The ceiling omits KV-cache traffic, compute limits, communication, and kernel overhead. Longer sequences increase the displayed memory but do not change this weight-only ceiling. +- Precision sets weight storage size. It does not establish that a GPU supports native arithmetic at that precision or that suitable kernels exist. +- Memory fit reserves 15% headroom. A configuration that needs sharding cannot achieve the displayed ceiling on a single GPU; multi-GPU performance is not modeled. +- Use measured benchmarks for deployment decisions. + +See [How To Scale Your Model: Transformer Inference](https://jax-ml.github.io/scaling-book/inference/) for the full bandwidth and compute model. diff --git a/src/content/tools/tokenizer-explorer/index.mdx b/src/content/tools/tokenizer-explorer/index.mdx index e057a36..1ff278e 100644 --- a/src/content/tools/tokenizer-explorer/index.mdx +++ b/src/content/tools/tokenizer-explorer/index.mdx @@ -1,27 +1,20 @@ --- name: Tokenizer explorer -summary: Paste any text and see how seven popular tokenizers split it — BPE, WordPiece, SentencePiece, tiktoken, and friends, side by side. -tag: Experimental +summary: Paste any text and see how popular tokenizers split it, side by side, with token counts and cost per provider. +tag: Soon icon: ▤ authors: - - lchen + - dinesh topics: - inference tags: [tokenizer, bpe, sentencepiece] featured: true --- -## What it does +## What we are building -Tokenizers determine how language models see your text. The same prompt can produce wildly different token counts (and therefore cost, latency, and behavior) across model families. This tool runs your text through seven tokenizers in parallel and shows the splits aligned column-by-column. - -## What you'll see - -- Token-level color coding so you can spot where tokenizers diverge -- Total token count + character compression ratio per tokenizer -- A "rare token" view that highlights tokens that decode to multiple Unicode codepoints -- Side-by-side cost projection across major API providers +Tokenizers decide how a model sees your text. The same prompt can produce very different token counts, and therefore cost and latency, across model families. This tool will run your text through several tokenizers in the browser and show the splits aligned column by column, with a token count and a cost projection per provider. ## Status -Experimental — currently runs entirely in-browser via WASM ports of the underlying libraries. Expect occasional Unicode edge cases. +Coming soon. Not yet available. Until then, the external tools on the playground page cover tokenization well. diff --git a/src/content/tools/training-compute-calc/TrainingComputeCalc.tsx b/src/content/tools/training-compute-calc/TrainingComputeCalc.tsx new file mode 100644 index 0000000..6037a1a --- /dev/null +++ b/src/content/tools/training-compute-calc/TrainingComputeCalc.tsx @@ -0,0 +1,315 @@ +'use client'; + +import { useState } from 'react'; +import { Field, Stat, StatRow } from '@/components/playground/primitives'; +import { + CHINCHILLA_TOKENS_PER_PARAM, + GPU_PEAK, + chinchillaTokensT, + formatCompact, + formatSci, + gpuHours, + impliedMfu, + tokensPerParam, + trainingFlops, + wallClockDays, + type GpuKey, +} from './compute'; + +type Preset = { + label: string; + params: number; + tokens: number; + gpu: GpuKey; + gpuExp?: number; + // GPU-hours the paper or model card reports, when one exists. + reportedGpuHours?: number; + reportedBy?: string; + mixedPrecision?: boolean; +}; + +const PRESETS: Preset[] = [ + { label: 'Chinchilla 70B', params: 70, tokens: 1.4, gpu: 'a100' }, + { + label: 'Llama 2 70B', + params: 70, + tokens: 2.0, + gpu: 'a100', + reportedGpuHours: 1_720_320, + reportedBy: 'Llama 2 model card', + }, + { label: 'Llama 3 8B', params: 8, tokens: 15, gpu: 'h100' }, + { + label: 'DeepSeek-V3 (37B active)', + params: 37, + tokens: 14.8, + gpu: 'h100', + gpuExp: 11, + reportedGpuHours: 2_664_000, + reportedBy: 'DeepSeek-V3 report, FP8 pretraining on H800s', + mixedPrecision: true, + }, + { + label: 'Llama 3 405B', + params: 405, + tokens: 15.6, + gpu: 'h100', + gpuExp: 14, + reportedGpuHours: 30_840_000, + reportedBy: 'Llama 3.1 model card, H100 training total', + }, +]; + +export default function TrainingComputeCalc({ compact = false }: { compact?: boolean }) { + const [params, setParams] = useState(70); + const [tokens, setTokens] = useState(1.4); + const [gpu, setGpu] = useState('h100'); + const [gpuExp, setGpuExp] = useState(10); + const [mfu, setMfu] = useState(40); + const [rate, setRate] = useState(2); + // The preset stays selected until another preset is chosen; sliders never clear it. + const [presetLabel, setPresetLabel] = useState(null); + + const gpuCount = 2 ** gpuExp; + const activePreset = PRESETS.find((p) => p.label === presetLabel); + // The paper's GPU-hours only compare cleanly while the model still matches the paper. + const presetIntact = + activePreset !== undefined && activePreset.params === params && activePreset.tokens === tokens; + const spec = GPU_PEAK[gpu]; + const flops = trainingFlops(params, tokens); + const hours = gpuHours(flops, spec.tflops, mfu / 100); + const days = wallClockDays(hours, gpuCount); + const ratio = tokensPerParam(params, tokens); + const sci = formatSci(flops); + const cost = hours * rate; + const effectiveTflops = spec.tflops * (mfu / 100); + const clusterPflops = (effectiveTflops * gpuCount) / 1000; + + const ratioNote = + ratio < CHINCHILLA_TOKENS_PER_PARAM * 0.75 + ? 'below the dense-model 20× reference' + : ratio > CHINCHILLA_TOKENS_PER_PARAM * 1.5 + ? 'above the dense-model 20× reference' + : 'near the dense-model 20× reference'; + + return ( +
+ {!compact && ( + <> +
+

+ Training Compute Calculator +

+ + · LIVE + +
+

+ How many FLOPs, GPU-hours, and days a pretraining run needs, from the 6ND rule and the + hardware you point at it. +

+ + )} + +
+ {PRESETS.map((p) => ( + + ))} +
+ +
+ + setParams(+e.target.value)} + style={{ width: '100%' }} + /> + + + setTokens(+e.target.value)} + style={{ width: '100%' }} + /> + + + + setGpuExp(+e.target.value)} + style={{ width: '100%' }} + /> + + + setMfu(+e.target.value)} + style={{ width: '100%' }} + /> + + +
+ {(Object.keys(GPU_PEAK) as GpuKey[]).map((k) => ( + + ))} +
+
+ + setRate(+e.target.value)} + style={{ width: '100%' }} + /> + +
+ +
+
+ Estimate +
+
+ + + = 1 ? `${days.toFixed(1)} days` : `${(days * 24).toFixed(1)} hours`} + /> +
+
+ + + + + + {presetIntact && activePreset?.reportedGpuHours && ( + + )} +
+
+ Approximate model compute using 6ND and dense BF16 peak throughput. MFU must reflect your + workload and cluster size; the estimate does not check memory fit or predict scaling + efficiency. Attention compute is omitted. Published GPU-hours are reference totals, not + predictions at the selected MFU. + {presetIntact && activePreset?.mixedPrecision && ( + <> + {' '} + DeepSeek used FP8 mixed precision on H800s. The selected H100 is a compute proxy; this + BF16 estimate is not a reproduction of that run. + + )} +
+
+
+ ); +} diff --git a/src/content/tools/training-compute-calc/compute.test.ts b/src/content/tools/training-compute-calc/compute.test.ts new file mode 100644 index 0000000..2a79bea --- /dev/null +++ b/src/content/tools/training-compute-calc/compute.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest'; +import { + chinchillaTokensT, + formatSci, + gpuHours, + impliedMfu, + tokensPerParam, + trainingFlops, + wallClockDays, +} from './compute'; + +describe('training compute math', () => { + it('reproduces the Llama 3 405B budget from the paper (3.8e25 FLOPs)', () => { + const flops = trainingFlops(405, 15.6); + expect(flops / 3.8e25).toBeCloseTo(1, 1); + }); + + it('converts 405B compute to hours at an assumed 34.6% utilization', () => { + const hours = gpuHours(trainingFlops(405, 15.6), 989, 0.346); + expect(hours / 30.84e6).toBeCloseTo(1, 1); + }); + + it('calculates the BF16 reference ratio from Llama 2 reported hours', () => { + const mfu = impliedMfu(trainingFlops(70, 2.0), 312, 1_720_320); + expect(mfu).toBeGreaterThan(0.4); + expect(mfu).toBeLessThan(0.47); + }); + + it('inverts GPU-hour accounting without treating the inferred ratio as a measurement', () => { + const flops = trainingFlops(37, 14.8); + const mfu = impliedMfu(flops, 989, 2_664_000); + expect(mfu).toBeGreaterThan(0.3); + expect(mfu).toBeLessThan(0.4); + const days = wallClockDays(gpuHours(flops, 989, mfu), 2048); + expect(days / (14.8 * 3.7)).toBeCloseTo(1, 1); + }); + + it('gives Chinchilla 70B its 1.4T tokens', () => { + expect(chinchillaTokensT(70)).toBeCloseTo(1.4, 5); + expect(tokensPerParam(70, 1.4)).toBeCloseTo(20, 5); + }); + + it('keeps the 20x token budget positive and exact across the parameter slider range', () => { + for (let params = 1; params <= 1000; params++) { + const tokens = chinchillaTokensT(params); + expect(tokens).toBeGreaterThanOrEqual(0.01); + expect(tokens).toBeLessThanOrEqual(30); + expect(tokensPerParam(params, tokens)).toBeCloseTo(20, 10); + } + expect(chinchillaTokensT(1)).toBe(0.02); + expect(chinchillaTokensT(3)).toBe(0.06); + }); + + it('converts a hand-calculated compute budget into GPU hours', () => { + // 1B parameters and 1T tokens = 6e21 FLOPs. + // 100 TFLOPS at 50% supplies 5e13 FLOPs/s: 120M seconds. + expect(trainingFlops(1, 1)).toBe(6e21); + expect(gpuHours(6e21, 100, 0.5)).toBeCloseTo(33333.333333, 5); + }); + + it('halves time with double utilization or double GPUs, holding other inputs fixed', () => { + const work = trainingFlops(70, 1.4); + const hours = gpuHours(work, 989, 0.4); + expect(gpuHours(work, 989, 0.2)).toBeCloseTo(hours * 2); + expect(wallClockDays(hours, 2048)).toBe(wallClockDays(hours, 1024) / 2); + }); + + it('distinguishes the full reported 405B hours from the 54-day snapshot', () => { + expect(wallClockDays(30_840_000, 16_384)).toBeCloseTo(78.4302, 4); + }); + + it('spreads GPU-hours across the cluster', () => { + expect(wallClockDays(24_000, 1000)).toBe(1); + }); + + it('formats scientific notation', () => { + expect(formatSci(3.79e25)).toEqual({ mantissa: '3.79', exponent: 25 }); + expect(formatSci(0)).toEqual({ mantissa: '0', exponent: 0 }); + }); +}); diff --git a/src/content/tools/training-compute-calc/compute.ts b/src/content/tools/training-compute-calc/compute.ts new file mode 100644 index 0000000..d77e945 --- /dev/null +++ b/src/content/tools/training-compute-calc/compute.ts @@ -0,0 +1,52 @@ +// Peak dense BF16 tensor-core throughput from NVIDIA datasheets (no 2:4 sparsity). +export const GPU_PEAK = { + a100: { name: 'A100 80GB', tflops: 312 }, + h100: { name: 'H100 SXM', tflops: 989 }, + h200: { name: 'H200 SXM', tflops: 989 }, + b200: { name: 'B200', tflops: 2250 }, +} as const; +export type GpuKey = keyof typeof GPU_PEAK; + +// Hoffmann et al. (Chinchilla, 2022): compute-optimal training uses ~20 tokens per parameter. +export const CHINCHILLA_TOKENS_PER_PARAM = 20; + +// Kaplan et al. (2020): forward + backward ≈ 6 FLOPs per parameter per token. +export function trainingFlops(paramsB: number, tokensT: number): number { + return 6 * paramsB * 1e9 * tokensT * 1e12; +} + +export function chinchillaTokensT(paramsB: number): number { + return (paramsB * 1e9 * CHINCHILLA_TOKENS_PER_PARAM) / 1e12; +} + +export function tokensPerParam(paramsB: number, tokensT: number): number { + return (tokensT * 1e12) / (paramsB * 1e9); +} + +export function gpuHours(flops: number, peakTflops: number, mfu: number): number { + return flops / (peakTflops * 1e12 * mfu) / 3600; +} + +export function wallClockDays(totalGpuHours: number, gpuCount: number): number { + return totalGpuHours / gpuCount / 24; +} + +// MFU as defined in the PaLM paper: achieved model FLOPs over peak hardware FLOPs. +export function impliedMfu(flops: number, peakTflops: number, reportedGpuHours: number): number { + return flops / (peakTflops * 1e12 * reportedGpuHours * 3600); +} + +export function formatSci(n: number): { mantissa: string; exponent: number } { + if (!isFinite(n) || n <= 0) return { mantissa: '0', exponent: 0 }; + const exponent = Math.floor(Math.log10(n)); + const mantissa = n / 10 ** exponent; + return { mantissa: mantissa.toFixed(2), exponent }; +} + +export function formatCompact(n: number): string { + if (!isFinite(n)) return '—'; + if (n >= 1e9) return `${(n / 1e9).toFixed(2)}B`; + if (n >= 1e6) return `${(n / 1e6).toFixed(2)}M`; + if (n >= 1e3) return `${(n / 1e3).toFixed(1)}K`; + return n.toFixed(0); +} diff --git a/src/content/tools/training-compute-calc/index.mdx b/src/content/tools/training-compute-calc/index.mdx new file mode 100644 index 0000000..0482e96 --- /dev/null +++ b/src/content/tools/training-compute-calc/index.mdx @@ -0,0 +1,55 @@ +--- +name: Training Compute Calculator +summary: Turn a model size and token budget into training FLOPs, GPU-hours, wall-clock days, and cost, using the 6ND rule and real GPU peak numbers. +tag: Live +icon: ∑ +authors: + - dinesh +topics: + - training + - distributed +tags: [training, flops, scaling-laws, chinchilla, mfu] +core: true +featured: true +--- + +import TrainingComputeCalc from './TrainingComputeCalc'; + + + +## What it does + +Pick a parameter count, a token budget, a GPU and a cluster size. The tool returns the total training compute, how many GPU-hours that is at a given utilization, how long the run takes on your cluster, and what it costs at your hourly rate. + +## The math + +- **Training FLOPs ≈ 6 × N × D**, where N is the parameter count and D is the training token count. The calculator converts billions of parameters and trillions of tokens to raw counts. This approximates the forward and backward parameter-matrix operations; it is not a complete operation count. +- **GPU-hours = FLOPs ÷ (peak FLOPS × MFU) ÷ 3600.** TFLOPS are multiplied by 10¹². MFU is entered as a percentage and converted to a fraction. +- **Days = GPU-hours ÷ GPU count ÷ 24.** This assumes the chosen MFU holds at the chosen cluster size. More GPUs do not automatically preserve utilization. +- **Cost = GPU-hours × hourly rate.** This is GPU rental cost only. + +The **20 tokens per parameter** button is a rough dense-model reference inspired by [Chinchilla](https://arxiv.org/abs/2203.15556), not a universal optimum. Data, architecture, and the intended inference budget change the best allocation. Applying that ratio to active MoE parameters does not establish compute optimality. + +[PaLM's MFU accounting](https://arxiv.org/abs/2204.02311) includes an additional attention term: approximately 12 × layers × query heads × head dimension × sequence length per token. This calculator omits it. The error depends on model size and context length; even at 8K context it can be substantial for smaller models. For example, 32 layers, 32 heads, head dimension 128, and 8,192 tokens add about 12.9 billion FLOPs per token, or 27% on top of 6N for an 8B model under that convention. + +## Published reference runs + +| Run | Parameters × tokens | Approximate 6ND FLOPs | Reported GPU-hours | +| -------------- | ------------------- | --------------------- | ----------------------------- | +| Llama 2 70B | 70B × 2T | 8.4 × 10²³ | 1,720,320 A100 | +| Llama 3.1 405B | 405B × 15.6T | 3.79 × 10²⁵ | 30.84M H100 | +| DeepSeek-V3 | 37B active × 14.8T | 3.29 × 10²⁴ | 2.664M H800, pretraining only | + +Sources: [Llama 2 model card](https://huggingface.co/meta-llama/Llama-2-70b), [Llama 3.1 model card](https://huggingface.co/meta-llama/Llama-3.1-405B), [Llama 3 report](https://arxiv.org/abs/2407.21783), and [DeepSeek-V3 report](https://arxiv.org/abs/2412.19437). + +Dividing approximate 6ND compute by reported GPU-hours and BF16 peak gives about 43.5% for Llama 2 70B and 34.5% for Llama 3.1 405B. These are inferred averages, not independent validation of the estimator or directly measured in-run MFU. Reported totals can cover different training stages and operating conditions. + +For scale, 30.84M GPU-hours divided by a constant 16,384 GPUs is 78.4 days. Meta's 54-day reliability observation is a snapshot, not the full training duration. DeepSeek's 2.664M pretraining GPU-hours divided by 2,048 GPUs is 54.2 days. Selecting a preset keeps your chosen MFU and hourly rate, so the estimate need not equal the reported total. + +## Hardware and limitations + +- Peak values use dense BF16 throughput, without structured sparsity: A100 312 TFLOPS, H100 SXM and H200 SXM approximately 989, and B200 2,250. See [NVIDIA A100](https://www.nvidia.com/en-us/data-center/a100/), [H100](https://www.nvidia.com/en-us/data-center/h100/), [H200](https://www.nvidia.com/en-us/data-center/h200/), and [NVIDIA's non-sparse throughput table](https://github.com/NVIDIA/exemplar-performance#gpu-peak-theoretical-throughput). +- The DeepSeek preset uses an H100 compute proxy for an H800 FP8 mixed-precision run. A BF16 denominator produces a _higher_, not lower, utilization percentage than a larger FP8 denominator for the same work and time. Mixed-precision utilization needs an explicit accounting convention, so this preset shows reported hours without an implied MFU comparison. +- For MoE, active parameters give only a rough compute estimate. Routing, attention, auxiliary objectives, and communication add work. Memory capacity still depends on total parameters and training state. +- Memory fit, batch size, parallelism, interconnect, failures, checkpointing, and data loading are not modeled independently. Choose an end-to-end MFU appropriate to those conditions; the calculator cannot determine whether the configuration is feasible. +- Rental cost excludes storage, networking, and separately billed failed or experimental runs. diff --git a/src/pages/about.astro b/src/pages/about.astro index 52e619d..75fd1f6 100644 --- a/src/pages/about.astro +++ b/src/pages/about.astro @@ -2,6 +2,7 @@ import BaseLayout from '@/layouts/BaseLayout.astro'; import PageIntro from '@/components/PageIntro.astro'; import FounderBee from '@/components/FounderBee.astro'; +import CommunityIllustration from '@/components/CommunityIllustration.astro'; import IconLinkCard from '@/components/IconLinkCard.astro'; import { SITE } from '@/lib/site'; @@ -26,7 +27,7 @@ const links = [
@@ -36,53 +37,70 @@ const links = [

-
-

- {SITE.name} is an open community of contributors — engineers, researchers, students, - and curious learners — writing down what we figure out about modern machine learning systems and - sharing it with the world. -

+
+
+

+ {SITE.name} is an open community of engineers, researchers, students, and curious + learners. We share what we learn about machine learning systems so others can build on it. +

-

- The goal is simple: collect honest, technically grounded writing about ML systems — the - kernels, the schedulers, the embeddings, the deployments — without the marketing gloss. - Anyone interested in this domain is welcome to write here, whether you've shipped systems - for years or you just figured something out last week. If you've learned something worth - passing on, come write with us. -

+

+ You'll find practical explanations and lessons from building ML systems, from kernels and + schedulers to training and deployment. We value clear writing, technical depth, and honest + accounts of what worked and what didn't. +

-

- We think this knowledge is about to matter much more than it does today — - here's why this exists. -

+

+ Whether you've built systems for years or just figured something out, your experience can + help someone else. Everyone is welcome to contribute. +

-

- You don't need an invitation to contribute. Open the editor, share an - explanation or experience, and submit it for review. Every article carries its author's - name; every useful perspective helps the community learn. -

+

+ Open the editor, share an explanation or experience, and submit it + for review. Every article is credited to its author. +

-

- Everything here is © its respective author, all rights reserved. Notes from the workbench. -

+

Why we're building this community.

- +

Content belongs to its respective author. All rights reserved.

-

- Created and maintained by Humble Bee. - -

+ + +

+ Created and maintained by Humble Bee. + +

+
+
diff --git a/src/pages/content-policy.astro b/src/pages/content-policy.astro index 2f8c958..3f5bd5b 100644 --- a/src/pages/content-policy.astro +++ b/src/pages/content-policy.astro @@ -1,5 +1,6 @@ --- import BaseLayout from '@/layouts/BaseLayout.astro'; +import PageIntro from '@/components/PageIntro.astro'; import { SITE } from '@/lib/site'; const canonical = `${SITE.url}/content-policy`; @@ -11,11 +12,8 @@ const canonical = `${SITE.url}/content-policy`; canonical={canonical} >
-
-
Policy
-

Content policy.

-

Last updated July 2026

-
+ +

Last updated July 2026

@@ -114,28 +112,17 @@ const canonical = `${SITE.url}/content-policy`; diff --git a/src/pages/index.astro b/src/pages/index.astro index 9241ff5..d8106cb 100644 --- a/src/pages/index.astro +++ b/src/pages/index.astro @@ -16,22 +16,14 @@ const reading = allPosts.filter((post) => post.id !== lead?.id).slice(0, 3); const counts = countPostsByTopic(allPosts); const topics = TOPICS.filter((topic) => counts[topic.id] > 0).slice(0, 3); const moreTopics = TOPICS.filter((topic) => !topics.some((featured) => featured.id === topic.id)); -const liveTools = rankTools( - await getCollection('tools', ({ data }) => !data.draft && data.tag === 'Live'), -); -const tools = pickCoreTools(liveTools, 2); -const moreTools = liveTools.filter((tool) => !tools.some((featured) => featured.id === tool.id)); -const toolDescriptions: Record = { - 'attention-viz': - 'Explore a token-by-token attention map, from causal masking to recurring attention patterns.', - 'throughput-calc': - 'Estimate memory-bound token throughput and KV-cache size across models, precisions, batch sizes, and GPUs.', -}; +const allTools = rankTools(await getCollection('tools', ({ data }) => !data.draft)); +const tools = pickCoreTools(allTools, 4); +const moreTools = allTools.filter((tool) => !tools.some((featured) => featured.id === tool.id)); ---

@@ -102,12 +94,7 @@ const toolDescriptions: Record = { {post.data.readMin} min read

- - {post.data.title} - - + {post.data.title}

{topic.desc}

- - {counts[topic.id]} {counts[topic.id] === 1 ? 'article' : 'articles'} - - + {counts[topic.id]} {counts[topic.id] === 1 ? 'article' : 'articles'} ))} @@ -178,7 +162,7 @@ const toolDescriptions: Record = { tools.length > 0 && (
-

ML systems, hands-on.

+

ML Systems tools

All tools @@ -193,7 +177,7 @@ const toolDescriptions: Record = {

{tool.data.name}

-

{toolDescriptions[tool.id] ?? tool.data.summary}

+

{tool.data.summary}

))} @@ -203,7 +187,12 @@ const toolDescriptions: Record = {

More to try

@@ -222,11 +211,13 @@ const toolDescriptions: Record = {

A useful explanation. A hard-won lesson. A question worth exploring. If you're learning or building ML systems, there's a place for your perspective here. -

Write with us Meet the community +

@@ -468,19 +459,7 @@ const toolDescriptions: Record = { line-height: 1.3; margin: 10px 0 12px; } - .reading-item h3 a { - display: flex; - justify-content: space-between; - align-items: baseline; - gap: 20px; - } - .reading-arrow { - font-family: var(--font-sans); - color: var(--ink-3); - font-size: 18px; - } - .topic-section:has(.publication-more-topics), - .tools-section:has(.publication-more-tools) { + .topic-section:has(.publication-more-topics) { padding-bottom: calc(var(--section-space) - 22px); } .topic-section .publication-more-topics, @@ -530,24 +509,85 @@ const toolDescriptions: Record = { line-height: 1.6; } .topic-end { - display: flex; - gap: 30px; - align-items: center; font-size: 12px; color: var(--ink-3); - } - .topic-end > :last-child { - font-size: 20px; + white-space: nowrap; } .publication-tools { display: grid; - grid-template-columns: 1fr 1fr; - gap: 64px; + grid-template-columns: repeat(2, minmax(0, 1fr)); + gap: 0 48px; + } + .tools-section { + position: relative; + isolation: isolate; + margin-top: 32px; + padding: clamp(40px, 4vw, 56px) 0 32px; + } + .tools-section::before { + content: ''; + position: absolute; + inset: 0 -24px; + z-index: -1; + pointer-events: none; + background: color-mix(in srgb, var(--accent) 3%, var(--paper)); + } + .tools-section .publication-section-head { + display: grid; + grid-template-columns: 1fr auto 1fr; + align-items: center; + margin-bottom: 40px; + } + .tools-section .publication-section-head h2 { + grid-column: 2; + text-align: center; + max-width: none; + } + .tools-section .publication-section-head > a { + justify-self: end; } .publication-tool { + position: relative; + isolation: isolate; display: flex; align-items: flex-start; gap: 24px; + padding: 24px 0; + } + .publication-tool::before { + content: ''; + position: absolute; + inset: 8px -12px; + z-index: -1; + border: 1px solid color-mix(in srgb, var(--accent) 35%, var(--line)); + border-radius: 10px; + opacity: 0; + pointer-events: none; + transition: opacity 160ms ease; + } + .publication-tool:hover::before, + .publication-tool:focus-visible::before { + opacity: 1; + } + .publication-tool:nth-child(-n + 2)::before { + top: -12px; + } + .publication-tool:nth-last-child(-n + 2)::before { + bottom: -12px; + } + @media (prefers-reduced-motion: reduce) { + .publication-tool::before { + transition: none; + } + } + .publication-tool:nth-child(-n + 2) { + padding-top: 0; + } + .publication-tool:nth-child(n + 3) { + border-top: 1px solid var(--line); + } + .publication-tool:nth-last-child(-n + 2) { + padding-bottom: 0; } .publication-tool-icon { flex: 0 0 56px; @@ -573,11 +613,29 @@ const toolDescriptions: Record = { } .publication-invitation { border-top: 1px solid var(--line-2); - padding: var(--section-space) 0; + position: relative; + isolation: isolate; + padding: clamp(64px, 7vw, 96px) 24px; display: grid; - grid-template-columns: 1.15fr 1fr; - gap: 64px; - align-items: center; + justify-items: center; + gap: 24px; + text-align: center; + } + .publication-invitation::before { + content: ''; + position: absolute; + inset: 0; + z-index: -1; + pointer-events: none; + background: radial-gradient( + ellipse at 50% 48%, + color-mix(in srgb, var(--accent) 9%, transparent), + color-mix(in srgb, var(--accent) 3%, transparent) 45%, + transparent 75% + ); + } + .publication-invitation .publication-label { + justify-content: center; } .publication-invitation h2 { font-family: var(--font-display); @@ -589,23 +647,28 @@ const toolDescriptions: Record = { } .invitation-copy p { color: var(--ink-2); - font-size: 18px; + font-size: 16px; line-height: 1.7; - max-width: 420px; - margin: 0 0 22px; + max-width: 540px; + margin: 0 auto 24px; + } + .invitation-actions { + display: flex; + align-items: center; + justify-content: center; + flex-wrap: wrap; + gap: 12px 28px; } .invitation-copy .publication-small-link { display: flex; width: fit-content; - margin-top: 12px; + margin-top: 0; } @media (max-width: 1000px) { .publication-hero { gap: 24px; } - .reading-layout, - .publication-tools, - .publication-invitation { + .reading-layout { gap: 36px; } .lead-story { @@ -618,6 +681,17 @@ const toolDescriptions: Record = { } } @media (max-width: 720px) { + .tools-section .publication-section-head { + grid-template-columns: 1fr; + gap: 4px; + margin-bottom: 28px; + } + .tools-section .publication-section-head h2 { + grid-column: 1; + } + .tools-section .publication-section-head > a { + justify-self: center; + } .publication-hero { grid-template-columns: 1fr; padding: 28px 0 var(--section-space); @@ -645,8 +719,7 @@ const toolDescriptions: Record = { gap: 6px; } .reading-layout, - .publication-tools, - .publication-invitation { + .publication-tools { grid-template-columns: 1fr; gap: 36px; } @@ -661,8 +734,7 @@ const toolDescriptions: Record = { .reading-item h3 { font-size: 23px; } - .topic-section:has(.publication-more-topics), - .tools-section:has(.publication-more-tools) { + .topic-section:has(.publication-more-topics) { padding-bottom: calc(var(--section-space) - 16px); } .topic-section .publication-more-topics, @@ -676,15 +748,25 @@ const toolDescriptions: Record = { .publication-topic h3 { font-size: 24px; } - .topic-end { - gap: 12px; - } - .topic-end > :first-child { - display: none; - } .publication-tool { gap: 20px; } + .publication-tools { + gap: 0; + } + .publication-tool:nth-child(n + 2) { + padding-top: 24px; + border-top: 1px solid var(--line); + } + .publication-tool:not(:last-child) { + padding-bottom: 24px; + } + .publication-tool:nth-child(n + 2)::before { + top: 8px; + } + .publication-tool:not(:last-child)::before { + bottom: 8px; + } .publication-tool h3 { font-size: 19px; } diff --git a/src/pages/playground/[tool].astro b/src/pages/playground/[tool].astro index 2b0f33f..0281c3d 100644 --- a/src/pages/playground/[tool].astro +++ b/src/pages/playground/[tool].astro @@ -106,6 +106,7 @@ const breadcrumbJsonLd = { } .tool-meta { display: flex; + flex-wrap: wrap; align-items: center; gap: 12px; margin-bottom: 24px; diff --git a/src/pages/playground/index.astro b/src/pages/playground/index.astro index a643062..321abf0 100644 --- a/src/pages/playground/index.astro +++ b/src/pages/playground/index.astro @@ -15,7 +15,7 @@ const coreTools = pickCoreTools(allTools).map((t) => ({ name: t.data.name, desc: t.data.summary, tag: t.data.tag, - available: t.data.tag === 'Live', + available: t.data.tag !== 'Soon', })); const catalogTools = rankTools(allTools); diff --git a/src/pages/privacy.astro b/src/pages/privacy.astro index 5456431..b2dbc99 100644 --- a/src/pages/privacy.astro +++ b/src/pages/privacy.astro @@ -1,5 +1,6 @@ --- import BaseLayout from '@/layouts/BaseLayout.astro'; +import PageIntro from '@/components/PageIntro.astro'; import { SITE } from '@/lib/site'; const canonical = `${SITE.url}/privacy`; @@ -11,11 +12,8 @@ const canonical = `${SITE.url}/privacy`; canonical={canonical} >
-
-
Policy
-

Privacy.

-

Last updated July 2026

-
+ +

Last updated September 2026

@@ -25,10 +23,11 @@ const canonical = `${SITE.url}/privacy`;

No accounts, no logins

- You can read everything without signing in. The in-browser writing tool at /write runs entirely in your browser — what you type stays on your device, and nothing is sent to us - until you choose to download it and submit it by pull request, issue, or email. + You can read everything without signing in. Drafts in the writing tool at /write + are saved on your device. When you submit an article for review, its content, attached files, + and author details are sent to our publishing service to create a public pull request on GitHub. + You can also download your draft and submit it yourself by pull request, issue, or email.

Analytics

@@ -82,27 +81,16 @@ const canonical = `${SITE.url}/privacy`; diff --git a/src/pages/tags/[tag]/[...page].astro b/src/pages/tags/[tag]/[...page].astro index 1097090..66a4515 100644 --- a/src/pages/tags/[tag]/[...page].astro +++ b/src/pages/tags/[tag]/[...page].astro @@ -3,6 +3,7 @@ import type { GetStaticPaths, Page } from 'astro'; import { getCollection } from 'astro:content'; import BaseLayout from '@/layouts/BaseLayout.astro'; import Pager from '@/components/Pager.astro'; +import PageIntro from '@/components/PageIntro.astro'; import PostRow from '@/components/PostRow.astro'; import { sortPostsByDate, tagSlug } from '@/lib/data'; import { resolvePostAuthors, type PostWithAuthors } from '@/lib/posts'; @@ -79,16 +80,12 @@ const title = jsonLd={[collectionJsonLd, breadcrumbJsonLd]} >
-
-
-
Tag
-

#{label}

-

- {page.total} - {page.total === 1 ? 'article' : 'articles'} tagged {label}. -

-
-
+ +

+ {page.total} + {page.total === 1 ? 'article' : 'articles'} tagged {label}. +

+
{page.data.map((a) => )} diff --git a/src/pages/why.astro b/src/pages/why.astro index bf51875..32927b5 100644 --- a/src/pages/why.astro +++ b/src/pages/why.astro @@ -10,7 +10,7 @@ const pageJsonLd = { '@type': 'WebPage', name: 'Why this exists', description: - 'Why learning machine learning systems matters: intelligence is moving to personal, local devices — and the systems layer decides how fast we get there.', + 'Why learning machine learning systems matters: intelligence is moving to personal, local devices, and systems engineering helps make that possible.', url: canonical, isPartOf: { '@type': 'WebSite', name: SITE.name, url: SITE.url }, }; @@ -18,76 +18,68 @@ const pageJsonLd = {
- + +

+ Understanding the systems behind AI helps more people build, question, and improve them. +

+

- Every important technology ends up boring. Electricity, databases, GPS — miracles that - became plumbing. Machine intelligence is on the same path, and we are living through its - plumbing years. This site is about those years, and the people doing the work. + ML systems turn model capabilities into something people can use. This site is for the + people learning how that happens and sharing what they discover.

-

Most "model progress" is systems progress

+

Systems make models useful

- Ask what actually changed between the demo that amazed you and the product you use every - day: tokens got cheaper, first tokens got faster, contexts got longer, models started - fitting on hardware you own. Almost none of that came from smarter weights. It came from - quantization, batching, caches, kernels, schedulers — the unglamorous layer underneath. - The distance between a demo and a product is measured in milliseconds and megabytes, and - systems engineers are the ones who close it. + A capable model is only part of a working product. Memory use, response time, training + cost, and reliability matter too. Quantization, batching, caches, kernels, and + schedulers help make models practical. We want to make that work easier to understand.

-

Computing always moves closer to you

+

More choice in where AI runs

- Mainframe to desktop, desktop to pocket, cloud to edge — every generation of computing - ends up nearer to the person using it, because latency, cost, and privacy all pull the - same way. Intelligence is on the same road. The endpoint is a model that runs on devices - you own, tuned on your own context — your notes, your work, your family's routines — a - private intelligence layer that answers to you and no one else. A model that knows you - that well shouldn't live in someone else's building. The datacenter era of AI is its - mainframe era, and the people who understand inference at the edge are the ones who will - end it. + Some workloads belong in a datacenter. Others benefit from running on a laptop, phone, + or device nearby. Local inference can offer privacy, offline access, and lower latency, + with real limits on memory and power. Understanding those tradeoffs gives people more + control over the systems they use.

-

The fundamentals outlast the headlines

+

Fundamentals outlast the headlines

- Architectures churn monthly; the systems layer barely moves. Memory hierarchies, - arithmetic intensity, batching tradeoffs, the cost of moving a byte versus computing on - it — these were true before transformers and will be true after them. Learning ML - systems is learning the invariants: knowledge that compounds for decades while the - leaderboards reshuffle. + Models and frameworks change quickly. Memory hierarchies, arithmetic intensity, + batching, and the cost of moving data remain useful ways to reason about them. Learning + these fundamentals helps you evaluate new ideas instead of starting from scratch each + time.

-

The bottleneck is people

+

Knowledge grows when we share it

- The knowledge that makes all of this work is concentrated in a handful of infrastructure - teams and scattered across conference talks and half-finished blog posts. That scarcity - is the real constraint on how fast the local, personal future arrives. The fix is old - and reliable: write things down, in the open, where anyone can learn them. A field grows - exactly as fast as its commons. + Useful knowledge is scattered across papers, code, talks, and individual experience. A + clear explanation or an honest account of a failed approach can save someone else days + of work. Publishing it openly makes that experience available beyond one team.

-

So we write

+

A place to contribute

- Articles, primers, and tools from practitioners — honest, technically grounded, free to - read, open to anyone who has figured something out and is willing to pass it on. If the - future we described sounds right to you, help build the commons that gets us there - sooner. + We bring together articles, primers, and tools from people learning and building ML + systems. Everything is free to read, and anyone can submit work for review. If you've + learned something worth passing on, there's room for it here.

@@ -194,14 +186,14 @@ const pageJsonLd = { } .why-lede { font-family: var(--font-read); - font-style: italic; + font-style: normal; font-size: 21px; line-height: 1.6; color: var(--ink); margin: 8px 0 0; } .why-body section { - margin-top: 48px; + margin-top: 36px; } .why-body h2 { font-family: var(--font-display); @@ -221,6 +213,6 @@ const pageJsonLd = { display: flex; gap: 12px; flex-wrap: wrap; - margin-top: 56px; + margin-top: 36px; } diff --git a/src/styles/publication.css b/src/styles/publication.css index 24deafd..83bb3f3 100644 --- a/src/styles/publication.css +++ b/src/styles/publication.css @@ -131,6 +131,9 @@ main { padding-top: 56px; } .footer-tagline { + font-family: var(--font-sans); + font-style: normal; + font-weight: 400; max-width: 320px; font-size: 15px; line-height: 1.7; @@ -597,32 +600,66 @@ input[type='radio'] { } .publication-topic-chips { display: flex; - gap: 10px; - overflow-x: auto; - padding: 3px 3px 6px; - scrollbar-width: thin; - scrollbar-color: var(--line-2) transparent; + flex-wrap: wrap; + align-items: baseline; + font-family: var(--font-read); + font-size: 17px; + letter-spacing: -0.01em; + color: var(--ink-2); } .publication-topic-chips a { - flex: 0 0 auto; - padding: 9px 16px; - border: 1px solid var(--line); - border-radius: 999px; - color: var(--ink-2); - font-size: 14px; - line-height: 1.4; + position: relative; + color: inherit; + padding: 2px 0; text-decoration: none; - transition: - border-color 150ms, - color 150ms; + transition: color 150ms; +} +.publication-topic-chips a::before { + content: ''; + position: absolute; + inset: -4px -7px; + border: 1px solid color-mix(in srgb, var(--accent) 35%, var(--line)); + border-radius: 6px; + opacity: 0; + pointer-events: none; + transition: opacity 160ms ease; +} +.publication-topic-chips a:not(:last-child) { + margin-right: 32px; +} +.publication-topic-chips a:not(:last-child)::after { + content: '·'; + position: absolute; + right: -18px; + color: var(--ink-4); +} +.publication-topic-chips a:hover::before, +.publication-topic-chips a:focus-visible::before { + opacity: 1; } .publication-topic-chips a:hover { - border-color: var(--accent); color: var(--accent); } .publication-topic-chips a:focus-visible { - outline: 2px solid var(--accent); - outline-offset: 1px; + outline: none; + color: var(--accent); +} +.publication-topic-chips a:focus-visible::before { + border-color: var(--accent); +} +@media (prefers-reduced-motion: reduce) { + .publication-topic-chips a, + .publication-topic-chips a::before { + transition: none; + } +} +.publication-chip-tag { + margin-left: 8px; + font: 10px/1 var(--font-mono); + letter-spacing: 0.08em; + text-transform: uppercase; + color: var(--ink-3); + vertical-align: middle; } .nav-topics { @@ -866,3 +903,12 @@ input[type='radio'] { transition: none; } } + +/* Keep schematic frames legible against the dark hero without changing data colors. */ +[data-theme='dark'] .hero-scenes { + --line: color-mix(in srgb, var(--ink-3) 48%, var(--paper)); + --line-2: color-mix(in srgb, var(--ink-3) 68%, var(--paper)); +} +[data-theme='dark'] .hero-scenes line[stroke='var(--ink-3)'][stroke-width='0.5'] { + stroke-width: 0.85; +}