Renderingen-US

Metal

Apple GPU compute shader source (MSL).

Part of the GPU lane. Every panel below is compiler output.

How to read it#

The same kernel, a different ABI. Compare it against the WGSL beside it — one Faber function, two unrelated shading languages, neither written by hand.

Measured support#

metal-text is a device-kernel emitter, not a general-language target. It lowers @ kernel compute kernels and related GPU views, and deliberately nothing else, so it is not scored against the general corpus — a percentage there would read as a completion score it is not. Its measured support is the device kernel support summary.

A compute kernel#

A function marked @ kernel. Device lanes only — this is a different kind of source, not a variant of the programs above.

Faber source

reader locale
faber convert --to en — English reader surface
@ kernel
fn multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → void {
    const tf32[16, 16] c ← a.matmul(b)
}
faber convert --to la — canonical Faber
@ nucleum
functio multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → vacuum {
    fixum tf32[16, 16] c ← a.matmul(b)
}
faber convert --to th-TH — Thai
@ เคอร์เนล
ฟังก์ชัน multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → เปล่า {
    คงที่ tf32[16, 16] c ← a.คูณเมทริกซ์(b)
}
faber convert --to zh-Hans — Simplified Chinese
@ 内核
函数 multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → 无值 {
    常量 tf32[16, 16] c ← a.矩阵乘法(b)
}
faber convert --to zh-Hant — Traditional Chinese
@ 內核
函式 multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → 空值 {
    定值 tf32[16, 16] c ← a.矩陣乘法(b)
}
faber convert --to vi — Vietnamese
@ hạt_nhân
hàm multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → trống {
    hằng tf32[16, 16] c ← a.nhân_ma_trận(b)
}
faber convert --to ar — Arabic
@ نواة
دالة multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → فراغ {
    ثابت tf32[16, 16] c ← a.ضرب_المصفوفات(b)
}
faber convert --to hi — Hindi
@ कर्नेल
फलन multiplico(tf32[16, 8] a, tf32[8, 16] b, tf32[16, 16] out, u32 id) → रिक्त {
    स्थिर tf32[16, 16] c ← a.आव्यूह_गुणन(b)
}

Metal — 4 lines in, 36 out (9.0×)

// Generated by radix metal-text (supported-with-limitations compute source).
#include <metal_stdlib>
using namespace metal;

kernel void multiplico(
    device const float* a_in [[buffer(0)]],
    device const float* b_in [[buffer(1)]],
    device float* output [[buffer(2)]],
    uint3 id [[thread_position_in_grid]],
    uint3 local_id [[thread_position_in_threadgroup]]
) {
    uint i = id.x;
threadgroup float shared_a[64];
threadgroup float shared_b[64];
float acc = 0.0;
uint row = id.y;
uint col = id.x;
uint ty = local_id.y;
uint tx = local_id.x;
for (uint k_tile = 0u; k_tile < 1u; k_tile++) {
    uint k_start = k_tile * 8u;
    uint a_idx = row * 8u + (k_start + tx);
    if (a_idx < 16u * 8u) { shared_a[ty * 8u + tx] = a_in[a_idx]; }
    if (a_idx >= 16u * 8u) { shared_a[ty * 8u + tx] = 0.0; }
    uint b_idx = (k_start + ty) * 16u + col;
    if (b_idx < 8u * 16u) { shared_b[ty * 8u + tx] = b_in[b_idx]; }
    if (b_idx >= 8u * 16u) { shared_b[ty * 8u + tx] = 0.0; }
    threadgroup_barrier(mem_flags::mem_threadgroup);
    for (uint kk = 0u; kk < 8u; kk++) {
        acc += shared_a[ty * 8u + kk] * shared_b[kk * 8u + tx];
    }
    threadgroup_barrier(mem_flags::mem_threadgroup);
}
uint out_idx = row * 16u + col;
if (row < 16u && col < 16u) { output[out_idx] = acc; }
}

---

All targets · Measured support per term