Repository navigation
413 lines (400 loc) · 20.2 KB
/
Copy pathsimd-matrix.yaml
File metadata and controls
413 lines (400 loc) · 20.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
on:
pull_request:
paths:
- 'src/simd*.rs'
- 'src/simd_nightly/**'
- 'src/simd_masking_ops.rs'
- 'src/hpc/amx_ops.rs'
- 'src/hpc/amx_matmul.rs'
- 'crates/simd-masking-parity/**'
- 'crates/neon-simd-parity/**'
- 'examples/ternlog_codegen_probe.rs'
- 'examples/amx_realization_report.rs'
- 'scripts/masking-parity.sh'
- 'scripts/codegen-witness.sh'
- 'scripts/neon-asm-rung3.sh'
- 'tools/gen_ternlog_bodies.py'
- '.cargo/**'
- 'Cargo.toml'
- 'src/lib.rs'
- '.github/workflows/simd-matrix.yaml'
merge_group:
push:
branches:
- master
- main
name: SIMD realization matrix
# Least privilege: nothing here pushes, comments, or releases. Every job
# only reads the tree, so the token is read-only and is not persisted into
# the checkout's git config (a repository-controlled command that runs after
# checkout would otherwise inherit whatever the repo's default token can do).
permissions:
contents: read
# Two axes, one program.
#
# realization × platform
# avx512 / avx2 / neon / wasm / scalar / nightly × x86_64 / aarch64 / wasm32
#
# Four rows run the SAME facade-only parity program unconditionally
# (`crates/simd-masking-parity`, via `scripts/masking-parity.sh <arm>`) — it
# has no idea which backend `simd.rs` selected, so a row proves "this
# realization is bit-identical to its scalar / bit-serial references" and
# nothing else. The avx512 row runs it ONLY on a runner that has avx512f;
# otherwise it degrades to an assembly-only assertion, and on that path the
# AVX-512 realization's bits are NOT proven in CI (the local v4 gate is the
# record for them). Where a row's assembly can be inspected, the tiny opt-3
# codegen oracle (`examples/ternlog_codegen_probe.rs`, via
# `scripts/codegen-witness.sh <arm>`) runs beside it: the parity program
# proves bits at the parity crate's release opt-level 2, the oracle proves the
# backend selected the instruction it is REQUIRED to select at opt-level 3.
# Neither replaces the other.
#
# The scalar realization has no host of its own: it is what `simd.rs` selects
# on wasm32 WITHOUT `+simd128`, so the `scalar` row is a wasm32 build with the
# feature off, run under node. `nightly` is the `core::simd` realization
# behind the opt-in `nightly-simd` feature and needs a nightly rustc.
#
# No workflow-global RUSTFLAGS here, on purpose: a global RUSTFLAGS REPLACES
# every cargo-config `rustflags` entry, which is how the v4 row would silently
# become a v3 row (see `.github/workflows/ci.yaml` tier4 for the incident).
# The v4 row passes `--config .cargo/config-v4.toml` through CARGO_ARGS.
env:
CARGO_TERM_COLOR: always
jobs:
native:
# x86_64 at x86-64-v3 (the AVX2 realization), PINNED EXPLICITLY via
# `.cargo/config-v3.toml`.
#
# ⊘ This row used to rely on v3 being `.cargo/config.toml`'s DEFAULT. That
# default is now `target-cpu=native`, so the pin is load-bearing rather
# than decorative, and the reason is two-sided-measured (2026-09-16, on an
# AVX-512 host): `codegen-witness.sh avx2` BARE reports
# "FAIL: ... has no packed logic" — it is grading vpternlog-carrying v4
# assembly against an assertion that says no vpternlog may appear — while
# the same command with `CARGO_ARGS='--config .cargo/config-v3.toml'`
# PASSES. Unpinned, this row would grade whichever tier the runner SKU
# happens to be; some Azure runner generations carry AVX-512, so it would
# be nondeterministic across reruns, not merely wrong.
#
# Also the ONLY row that can exercise AMX: the tile ops are
# runtime-gated and always compiled into native builds, so the report
# prints which gates this runner clears (`tile_available`/`available`
# false on a non-AMX runner is the expected, honest answer) and the
# encoding tests pin the assembled bytes without executing a tile op.
runs-on: ubuntu-latest
name: realization/avx2 × x86_64
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
- name: generated ternlog bodies are current
run: python3 tools/gen_ternlog_bodies.py --check
- name: masking parity (x86-64-v3, pinned)
run: CARGO_ARGS='--config .cargo/config-v3.toml' bash scripts/masking-parity.sh native
- name: codegen witness (avx2, pinned to v3)
run: CARGO_ARGS='--config .cargo/config-v3.toml' bash scripts/codegen-witness.sh avx2
- name: AMX realization report (runtime-gated; prints this runner's gates)
run: cargo run --example amx_realization_report
- name: AMX encoding + detection tests (no tile op executes)
run: cargo test --lib -- hpc::amx_ops simd_amx
host-native:
# INFORMATIONAL, non-gating (`continue-on-error`). Testing the waters:
# what does a GitHub runner actually GIVE us under the new
# `target-cpu=native` default?
#
# Every other row in this matrix names its tier and asserts against it.
# This one names NOTHING and just reports — because the open question the
# native default raises is empirical and we do not have the answer:
# GitHub's ubuntu-latest pool is not one SKU, and some generations carry
# AVX-512 while others do not. The parity program's own header line
# (`avx512f=true|false`) is the reading; `lscpu` beside it is the
# corroboration.
#
# ANSWER, measured 2026-09-16 on this row's first run — and it is stronger
# than the question asked. Within ONE workflow run (35148155422), two jobs
# both `runs-on: ubuntu-latest`, both under the `target-cpu=native`
# default, reported DIFFERENT tiers:
#
# realization/nightly x x86_64 avx512f=TRUE
# realization/host-native x x86_64 avx512f=FALSE
#
# So the pool is HETEROGENEOUS and the tier is decided per JOB, not per
# run and not per repo. (The nightly row's failure on the previous head
# was caused by landing on an AVX-512 runner, which compiled
# `#[cfg(all(test, target_feature = "avx512f"))]` modules that had never
# been compiled in CI before — a real polyfill gap, fixed in c1bd7015.)
#
# That is exactly why `native` pins nothing and the portable row pins v3:
# unpinned, an ISA assertion here would be a coin flip per job, and a
# green run would prove only that today's scheduling was lucky.
#
# It is `continue-on-error` ON PURPOSE and must stay that way: a row whose
# result is "whatever this runner is" cannot gate a merge without making
# the merge depend on pool scheduling. If a future session wants to ASSERT
# a tier here, that is a different row with an explicit pin — do not
# promote this one by deleting the flag.
#
# What it can still catch, and why it is worth a row at all: the parity
# program must be bit-identical to its scalar references on WHATEVER
# realization it lands on. A red here is a real parity failure on a tier
# no pinned row happens to cover.
runs-on: ubuntu-latest
name: realization/host-native × x86_64 (informational)
continue-on-error: true
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
# The build below resolves `target-cpu=native` on THIS runner, and the
# pool is heterogeneous (see host-native). A cache keyed only by job
# restores proc-macro .so files compiled for another runner's CPU, and
# rustc dies loading them with SIGILL (seen on ndarray #344: the cached
# `paste` .so). Key the cache by the target features rustc resolves for
# `native` here, which is exactly what decides those artifacts.
- name: native target features (cache key)
id: cpu
run: echo "features=$(rustc +stable --print cfg -C target-cpu=native | grep target_feature | sha256sum | cut -c1-16)" >> "$GITHUB_OUTPUT"
- uses: Swatinem/rust-cache@v2
with:
key: host-native-${{ steps.cpu.outputs.features }}
- name: what silicon is this runner
run: lscpu | sed -n '1,/^Flags/p' | head -30
- name: masking parity (host-native, unpinned — READ THE HEADER LINE)
run: bash scripts/masking-parity.sh native
native-v4:
# Same host, AVX-512 realization via the v4 cargo config. Building and
# inspecting the assembly needs no AVX-512 silicon; RUNNING the parity
# binary and the probe's self-check does, and GitHub's ubuntu runners do
# not promise it — so the run steps are gated on /proc/cpuinfo and report
# SKIPPED loudly rather than SIGILL. The build + witness inspection (which
# asserts vpternlog was selected) always runs.
runs-on: ubuntu-latest
name: realization/avx512 × x86_64
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
with:
key: v4
- name: detect the full x86-64-v4 AVX-512 set on this runner
id: cpu
# A v4 build may emit any of F/BW/CD/DQ/VL (the masking ops use the
# BW/VL byte and word compares beside the F ternlog), so avx512f alone
# would let a partially-capable host reach the run steps and SIGILL.
# All five or none.
run: |
has=1
for f in avx512f avx512bw avx512cd avx512dq avx512vl; do
grep -q -w "$f" /proc/cpuinfo || { echo "missing: $f"; has=0; }
done
echo "has=$has" >> "$GITHUB_OUTPUT"
grep -m1 'model name' /proc/cpuinfo || true
- name: build the parity program at x86-64-v4
run: env -u RUSTFLAGS cargo --config .cargo/config-v4.toml build --release --manifest-path crates/simd-masking-parity/Cargo.toml --bin simd-masking-parity --target x86_64-unknown-linux-gnu
- name: masking parity (native, v4 config)
if: steps.cpu.outputs.has == '1'
run: CARGO_ARGS='--config .cargo/config-v4.toml' bash scripts/masking-parity.sh native
- name: codegen witness (avx512) — assembly inspection + native self-check
if: steps.cpu.outputs.has == '1'
run: CARGO_ARGS='--config .cargo/config-v4.toml' bash scripts/codegen-witness.sh avx512
- name: codegen witness (avx512) — assembly inspection only (runner lacks avx512f)
if: steps.cpu.outputs.has == '0'
# The SAME script, in its asm-only mode — one implementation of the
# stale-assembly guard (`rm -f` + `touch`) and of the symbol
# attribution, not a second hand-rolled copy that drifts.
run: |
echo "::warning::runner lacks avx512f — v4 parity run and probe self-check SKIPPED; asserting the emitted assembly only"
env -u RUSTFLAGS WITNESS_NO_RUN=1 CARGO_ARGS='--config .cargo/config-v4.toml' bash scripts/codegen-witness.sh avx512
cpu-guard-library:
# `cpu_guard` must never end a process it does not own. A v4 cdylib that
# links ndarray, dlopen'ed by a C host under `qemu -cpu Haswell`, must
# warn and return; a v4 executable on the same emulated CPU must still
# exit 132. Before the library rule (2026-10-11) the cdylib ended its
# host with 132, which is what a JVM or Python process loading an
# ndarray-linked library would have suffered.
runs-on: ubuntu-latest
name: cpu_guard × library host
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
with:
key: cpu-guard-library
- name: install qemu-user
run: sudo apt-get update && sudo apt-get install -y qemu-user-static
- name: library warns, executable exits 132 (qemu -cpu Haswell)
run: bash scripts/cpu-guard-library-probe.sh
avx:
# AVX-without-AVX2 realization (`src/simd_avx.rs`, Sandy Bridge class).
# Built for `sandybridge` (`.cargo/config-avx.toml`) and RUN under
# `qemu-x86_64-static -cpu SandyBridge`, so an AVX2 instruction anywhere
# on the executed path faults instead of passing silently on an AVX2
# runner. Three rungs: parity (bits), an assembly witness that the parity
# binary contains no AVX2 integer instruction at all (LLVM never emits
# AVX2 for this target, so any `vp*` on a ymm register other than the
# AVX1 `vptest` came from an intrinsic that escaped its gate), and the
# SIMD unit tests under emulation.
runs-on: ubuntu-latest
name: realization/avx × x86_64
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
with:
key: avx
- name: install qemu-user
run: sudo apt-get update && sudo apt-get install -y qemu-user-static
- name: masking parity (sandybridge build, qemu -cpu SandyBridge)
run: bash scripts/masking-parity.sh avx-qemu
- name: no AVX2 instruction in the parity binary
run: |
bin=crates/simd-masking-parity/target/release/simd-masking-parity
# Sandy Bridge has AVX but neither AVX2 (integer ops on ymm), FMA3
# nor F16C. The parity binary has no runtime-gated paths, so any of
# them here came from an intrinsic the AVX arm failed to gate.
dis=$(objdump -d --no-show-raw-insn "$bin")
bad=$( { echo "$dis" | grep -E '\svp[a-z0-9]+\s.*%ymm' | grep -vE '\svptest\s';
echo "$dis" | grep -E '\sv(fn?m(add|sub)|fmaddsub|fmsubadd)[0-9a-z]*\s|\svcvt(ph2ps|ps2ph)\s'; } || true)
if [ -n "$bad" ]; then echo "$bad" | head -20; echo "::error::AVX2/FMA/F16C instruction in the AVX-arm parity binary"; exit 1; fi
echo "no AVX2, FMA or F16C instruction found"
- name: SIMD unit tests under qemu -cpu SandyBridge
env:
CARGO_PROFILE_DEV_DEBUG: "0"
CARGO_PROFILE_TEST_DEBUG: "0"
CARGO_PROFILE_DEV_OPT_LEVEL: "1"
CARGO_PROFILE_TEST_OPT_LEVEL: "1"
run: |
bin=$(env -u RUSTFLAGS cargo --config .cargo/config-avx.toml test --lib --no-run --message-format=json \
| jq -r 'select(.executable != null and .target.kind == ["lib"] and .target.name == "ndarray") | .executable')
# One thread: qemu-user itself has crashed (internal SIGSEGV) under
# the multithreaded harness while a test was panicking.
qemu-x86_64-static -cpu SandyBridge "$bin" simd --test-threads=1
v2:
# x86-64-v2 realization (SSE4.2, no AVX): `simd.rs` routes it to the
# scalar backend, which LLVM vectorizes to SSE. Built for `x86-64-v2`
# (`.cargo/config-v2.toml`) and RUN under `qemu-x86_64-static -cpu
# Nehalem`, so any AVX instruction on the executed path faults. Three
# rungs, as for `avx`: parity, an assembly witness that the parity binary
# has no AVX encoding at all, and the SIMD unit tests under emulation.
runs-on: ubuntu-latest
name: realization/v2 × x86_64
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
- uses: Swatinem/rust-cache@v2
with:
key: v2
- name: install qemu-user
run: sudo apt-get update && sudo apt-get install -y qemu-user-static
- name: masking parity (x86-64-v2 build, qemu -cpu Nehalem)
run: bash scripts/masking-parity.sh v2-qemu
- name: no AVX instruction in the parity binary
run: |
bin=crates/simd-masking-parity/target/release/simd-masking-parity
# Nehalem has no AVX: no ymm/zmm or mask register, and no VEX-encoded
# (v-prefixed) op on xmm either. The parity binary has no
# runtime-gated paths, so any hit came from an ungated intrinsic.
bad=$(objdump -d --no-show-raw-insn "$bin" | grep -E '%ymm|%zmm|%k[0-7]|\sv[a-z0-9]+\s.*%xmm' || true)
if [ -n "$bad" ]; then echo "$bad" | head -20; echo "::error::AVX instruction in the v2 parity binary"; exit 1; fi
echo "no AVX instruction found"
- name: SIMD unit tests under qemu -cpu Nehalem
env:
CARGO_PROFILE_DEV_DEBUG: "0"
CARGO_PROFILE_TEST_DEBUG: "0"
CARGO_PROFILE_DEV_OPT_LEVEL: "1"
CARGO_PROFILE_TEST_OPT_LEVEL: "1"
run: |
bin=$(env -u RUSTFLAGS cargo --config .cargo/config-v2.toml test --lib --no-run --message-format=json \
| jq -r 'select(.executable != null and .target.kind == ["lib"] and .target.name == "ndarray") | .executable')
# One thread: see the `avx` job.
qemu-x86_64-static -cpu Nehalem "$bin" simd --test-threads=1
neon:
# aarch64 realization: cross-build on the x86 runner, run under qemu-user.
# Three rungs: parity under qemu (bits), the codegen witness (opt-3 NEON
# logic on v*.16b, GPR logic bounded), and rung 3 of the pre-existing NEON
# asm gate (`neon-simd-parity`, the wider type surface).
runs-on: ubuntu-latest
name: realization/neon × aarch64
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
with:
targets: aarch64-unknown-linux-gnu
- run: rustup target add aarch64-unknown-linux-gnu
- uses: Swatinem/rust-cache@v2
with:
key: aarch64
- name: install aarch64 cross toolchain + qemu-user
run: sudo apt-get update && sudo apt-get install -y gcc-aarch64-linux-gnu qemu-user-static
- name: masking parity (neon, qemu)
run: bash scripts/masking-parity.sh neon-qemu
- name: codegen witness (neon)
run: CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER=aarch64-linux-gnu-gcc bash scripts/codegen-witness.sh neon aarch64-unknown-linux-gnu
- name: NEON asm rung 3
run: CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER=aarch64-linux-gnu-gcc bash scripts/neon-asm-rung3.sh
wasm:
# wasm32 with +simd128 = the `simd_wasm` realization; wasm32 WITHOUT it is
# what `simd.rs` selects as the scalar realization — the scalar backend's
# only executable row, run through the identical program.
runs-on: ubuntu-latest
name: realization/wasm + scalar × wasm32
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@stable
with:
targets: wasm32-unknown-unknown
- run: rustup target add wasm32-unknown-unknown
- uses: Swatinem/rust-cache@v2
with:
key: wasm
- uses: actions/setup-node@v4
with:
node-version: "22"
- name: masking parity (wasm, +simd128)
run: bash scripts/masking-parity.sh wasm
- name: masking parity (scalar realization = wasm32 without simd128)
run: bash scripts/masking-parity.sh wasm-scalar
nightly:
# The `core::simd` realization behind the opt-in `nightly-simd` feature.
# Same program, same reference, nightly rustc; plus the lib tests that
# exercise the arm directly (masking ops, facade tests, AMX encodings —
# the latter because nightly's newer LLVM is where a dropped mnemonic
# first surfaces, as TF32 did).
runs-on: ubuntu-latest
name: realization/nightly × x86_64
steps:
- uses: actions/checkout@v4
with:
persist-credentials: false
- uses: dtolnay/rust-toolchain@nightly
# The build below resolves `target-cpu=native` on THIS runner, and the
# pool is heterogeneous (see host-native). A cache keyed only by job
# restores proc-macro .so files compiled for another runner's CPU, and
# rustc dies loading them with SIGILL (seen on ndarray #344: the cached
# `paste` .so). Key the cache by the target features rustc resolves for
# `native` here, which is exactly what decides those artifacts.
- name: native target features (cache key)
id: cpu
run: echo "features=$(rustc +nightly --print cfg -C target-cpu=native | grep target_feature | sha256sum | cut -c1-16)" >> "$GITHUB_OUTPUT"
- uses: Swatinem/rust-cache@v2
with:
key: nightly-${{ steps.cpu.outputs.features }}
- name: masking parity (nightly-simd)
run: bash scripts/masking-parity.sh nightly
- name: lib tests on the nightly arm
run: cargo +nightly test --lib --features nightly-simd -- simd_masking_ops simd::tests hpc::amx_ops simd_amx