Spaces:
Running
Running
File size: 9,675 Bytes
13b1a91 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 | // Browser hardware estimate via WebGPU. This is NOT MLX and NOT LLM inference:
// it measures relative GPU compute in this browser, for coarse hardware grouping.
"use strict";
(function () {
const BENCH_VERSION = "webgpu-1";
// Reference throughputs that map to a score of 1000 in each phase. They are
// fixed constants so scores stay comparable across versions of this file.
const REF = { mm256: 30, mm512: 30, mm1024: 30, copy: 20 }; // GFLOPS, GFLOPS, GFLOPS, GB/s
const MATMUL_WGSL = `
struct Dims { n : u32 }
@group(0) @binding(0) var<storage, read> a : array<f32>;
@group(0) @binding(1) var<storage, read> b : array<f32>;
@group(0) @binding(2) var<storage, read_write> c : array<f32>;
@group(0) @binding(3) var<uniform> dims : Dims;
@compute @workgroup_size(16, 16)
fn main(@builtin(global_invocation_id) id : vec3<u32>) {
let n = dims.n;
if (id.x >= n || id.y >= n) { return; }
var s = 0.0;
for (var k = 0u; k < n; k = k + 1u) {
s = s + a[id.y * n + k] * b[k * n + id.x];
}
c[id.y * n + id.x] = s;
}`;
function browserFamily() {
const ua = navigator.userAgent;
if (/Edg\//.test(ua)) return "edge";
if (/Firefox\//.test(ua)) return "firefox";
if (/Chrome\//.test(ua)) return "chrome";
if (/Safari\//.test(ua)) return "safari";
return "other";
}
function osFamily() {
const p = (navigator.userAgentData && navigator.userAgentData.platform) || navigator.platform || "";
const ua = navigator.userAgent;
if (/iPhone|iPad|iPod/.test(ua)) return "ios";
if (/Mac/i.test(p)) return navigator.maxTouchPoints > 1 ? "ios" : "macos";
if (/Win/i.test(p)) return "windows";
if (/Android/i.test(ua)) return "android";
if (/Linux/i.test(p)) return "linux";
return "other";
}
// Keep only short, generic identifiers ("apple", "metal-3"), never full descriptions.
function token(s) {
return (typeof s === "string" && /^[A-Za-z0-9 ._+-]{1,24}$/.test(s)) ? s.toLowerCase() : null;
}
function memoryPrior(deviceMemory) {
// Chrome reports RAM rounded DOWN to a power of two and capped (currently 32).
// So 16 means "16 to 31 GB" and 32 means "32 GB or more".
if (!deviceMemory) return null;
if (deviceMemory >= 32) return { ram: 32, label: "32 GB+ class", note: "your browser reports at least 32 GB" };
if (deviceMemory >= 16) return { ram: 16, label: "16 GB-class (16 to 31 GB)", note: "your browser reports 16 GB or more" };
if (deviceMemory >= 8) return { ram: 8, label: "8 GB-class (8 to 15 GB)", note: "your browser reports 8 GB or more" };
return { ram: 8, label: "8 GB-class or less", note: "your browser reports under 8 GB" };
}
function capabilityClass(quickScore) {
if (quickScore == null) return "unknown";
if (quickScore >= 4000) return "high";
if (quickScore >= 1500) return "mid";
return "entry";
}
const CLASS_LABEL = {
high: "Higher-tier GPU compute",
mid: "Mid-tier GPU compute",
entry: "Entry-level GPU compute",
unknown: "GPU compute unknown",
};
async function getDevice() {
if (!("gpu" in navigator) || !navigator.gpu) return { error: "unavailable" };
let adapter;
try {
adapter = await navigator.gpu.requestAdapter({ powerPreference: "high-performance" });
} catch (e) {
return { error: "unavailable" };
}
if (!adapter) return { error: "unavailable" };
try {
const device = await adapter.requestDevice();
device.lost.then(() => {});
return { adapter, device };
} catch (e) {
return { error: "device", adapter };
}
}
function makeMatmul(device, n) {
const size = n * n * 4;
const mk = (usage) => device.createBuffer({ size, usage });
const a = mk(GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST);
const b = mk(GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST);
const c = mk(GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC);
const u = device.createBuffer({ size: 4, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST });
const data = new Float32Array(n * n);
for (let i = 0; i < data.length; i++) data[i] = ((i * 2654435761) % 1000) / 1000;
device.queue.writeBuffer(a, 0, data);
device.queue.writeBuffer(b, 0, data);
device.queue.writeBuffer(u, 0, new Uint32Array([n]));
const module = device.createShaderModule({ code: MATMUL_WGSL });
const pipeline = device.createComputePipeline({ layout: "auto", compute: { module, entryPoint: "main" } });
const bind = device.createBindGroup({
layout: pipeline.getBindGroupLayout(0),
entries: [a, b, c, u].map((buffer, i) => ({ binding: i, resource: { buffer } })),
});
const groups = Math.ceil(n / 16);
return {
async run(passes) {
const enc = device.createCommandEncoder();
for (let p = 0; p < passes; p++) {
const pass = enc.beginComputePass();
pass.setPipeline(pipeline);
pass.setBindGroup(0, bind);
pass.dispatchWorkgroups(groups, groups);
pass.end();
}
device.queue.submit([enc.finish()]);
await device.queue.onSubmittedWorkDone();
},
destroy() { [a, b, c, u].forEach((x) => x.destroy()); },
};
}
// Run matmul for ~ms milliseconds; returns GFLOPS (billions of multiply-adds per second).
async function timeMatmul(device, n, ms, onTick) {
const mm = makeMatmul(device, n);
try {
await mm.run(1); // warm-up + pipeline compile
let passes = 1, done = 0;
const t0 = performance.now();
while (performance.now() - t0 < ms) {
const s = performance.now();
await mm.run(passes);
done += passes;
if (performance.now() - s < 50) passes = Math.min(passes * 2, 4096);
if (onTick) onTick((performance.now() - t0) / ms);
}
const secs = (performance.now() - t0) / 1000;
return (done * n * n * n) / secs / 1e9;
} finally {
mm.destroy();
}
}
async function timeCopy(device, ms, onTick) {
const bytes = 256 * 1024 * 1024;
let src, dst;
try {
src = device.createBuffer({ size: bytes, usage: GPUBufferUsage.COPY_SRC | GPUBufferUsage.STORAGE });
dst = device.createBuffer({ size: bytes, usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.STORAGE });
} catch (e) {
return null;
}
try {
let copies = 0;
const t0 = performance.now();
while (performance.now() - t0 < ms) {
const enc = device.createCommandEncoder();
for (let i = 0; i < 4; i++) enc.copyBufferToBuffer(src, 0, dst, 0, bytes);
device.queue.submit([enc.finish()]);
await device.queue.onSubmittedWorkDone();
copies += 4;
if (onTick) onTick((performance.now() - t0) / ms);
}
const secs = (performance.now() - t0) / 1000;
return (copies * bytes) / secs / 1e9;
} finally {
src.destroy();
dst.destroy();
}
}
let cached = null;
async function detect() {
const facts = {
browser_family: browserFamily(),
os_family: osFamily(),
cpu_cores: Number.isInteger(navigator.hardwareConcurrency) ? navigator.hardwareConcurrency : null,
device_memory: typeof navigator.deviceMemory === "number" ? navigator.deviceMemory : null,
webgpu_available: false,
gpu_vendor: null,
gpu_arch: null,
quick_score: null,
capability: "unknown",
duration_ms: 0,
error: null,
};
const t0 = performance.now();
const got = await getDevice();
if (got.adapter && got.adapter.info) {
facts.gpu_vendor = token(got.adapter.info.vendor);
facts.gpu_arch = token(got.adapter.info.architecture);
}
if (!got.device) {
facts.error = got.error || "unavailable";
} else {
facts.webgpu_available = true;
try {
const gflops = await timeMatmul(got.device, 256, 1200);
facts.quick_score = Math.round((gflops / REF.mm256) * 1000);
facts.capability = capabilityClass(facts.quick_score);
} catch (e) {
facts.error = "test_failed";
}
cached = got.device;
}
facts.duration_ms = Math.round(performance.now() - t0);
facts.memory_prior = memoryPrior(facts.device_memory);
facts.capability_label = CLASS_LABEL[facts.capability];
return facts;
}
async function fullBenchmark(onProgress) {
const device = cached || (await getDevice()).device;
if (!device) throw new Error("WebGPU isn't available in this browser.");
const phases = [
["mm256", 4000, (t) => timeMatmul(device, 256, 4000, t)],
["mm512", 6000, (t) => timeMatmul(device, 512, 6000, t)],
["mm1024", 7000, (t) => timeMatmul(device, 1024, 7000, t)],
["copy", 3000, (t) => timeCopy(device, 3000, t)],
];
const total = phases.reduce((s, p) => s + p[1], 0);
let before = 0;
const raw = {};
const t0 = performance.now();
for (const [name, ms, fn] of phases) {
raw[name] = await fn((f) => onProgress && onProgress(Math.min(99, ((before + f * ms) / total) * 100), name));
before += ms;
}
const parts = Object.entries(raw).filter(([, v]) => v).map(([k, v]) => v / REF[k]);
const score = Math.round(Math.exp(parts.reduce((s, x) => s + Math.log(x), 0) / parts.length) * 1000);
if (onProgress) onProgress(100, "done");
return {
score,
raw: Object.fromEntries(Object.entries(raw).map(([k, v]) => [k, v && Math.round(v * 10) / 10])),
duration_ms: Math.round(performance.now() - t0),
version: BENCH_VERSION,
};
}
window.HW = { detect, fullBenchmark, BENCH_VERSION, capabilityClass };
})();
|