Skip to content

Benchmark sources

These are the programs behind the performance figures, each written four times with the same algorithm and the same data. Choose a language on any of them and every one switches to it.

The xc versions are plain loops over arrays and objects. None of them asks for vector instructions, threads, a matrix unit or anything else by name: where matrix_mul_f32 runs on Apple’s SME matrix unit, or array_map on AVX-512, it is because xcc saw the loop and chose to.

Each program times its own work with bench_now_us() (the monotonic clock) and prints a checksum, which must agree across the four languages for a run to count.

Allocate an object per iteration and let ARC free it.

Lines: xc 31, Objective-C 35, C++ 34, Swift 24.

// arc_alloc — allocate an object per iteration and let ARC free it.
// Exercises: heap allocation, retain/release traffic, method dispatch.
// Each object outlives its iteration (it is kept until the next one replaces
// it), so no compiler can put it on the stack or fold the loop away, and the
// result is folded in with ^ so the sum has no closed form.
#import "Stdio.xc"
#import "include/bench_time.xc"
class Node
{
u32 v;
void init(u32 x) { v = x; }
u32 get(void) { return v; }
}
i32 main(i32 argc, u8** argv)
{
u32 seed = (u32)argc;
u32 acc = (u32)0;
Node* keep = new Node(seed);
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)80000000; r++)
{
Node* n = new Node(r + seed);
acc = (acc ^ n.get()) + keep.get();
keep = n;
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Hold objects in an array and walk them.

Lines: xc 21, Objective-C 31, C++ 34, Swift 25.

// arc_array — hold objects in an array and walk them.
// Exercises: retain/release on stored references, field access through a pointer.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 1024
class Cell { u32 v; void init(u32 x) { v = x; } u32 get(void) { return v; } }
i32 main(i32 argc, u8** argv)
{
u32 seed = (u32)argc;
Cell* cells[N];
for (u32 i = (u32)0; i < (u32)N; i++) cells[i] = new Cell(i + seed);
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)3000000; r++)
for (u32 i = (u32)0; i < (u32)N; i++) acc = acc + cells[i].get();
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Elementwise c[i] = a[i] + b[i] * k.

Lines: xc 21, Objective-C 22, C++ 22, Swift 24.

// array_map — elementwise c[i] = a[i] + b[i] * k.
// Exercises: elementwise map loop, the shape the arm64 map vectoriser takes.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
u32 a[N]; u32 b[N]; u32 c[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++)
{ a[i] = i + seed; b[i] = (i * (u32)3) + seed; }
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)5000000; r++)
for (u32 i = (u32)0; i < (u32)N; i++)
c[i] = a[i] + (b[i] * (u32)7) + r;
u32 acc = (u32)0;
for (u32 i = (u32)0; i < (u32)N; i++) acc = acc + c[i];
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Sum an array. Exercises: reduction loop, array addressing.

Lines: xc 18, Objective-C 20, C++ 21, Swift 18.

// array_sum — sum an array. Exercises: reduction loop, array addressing.
// This is the shape the arm64 reduction vectoriser recognises.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
u32 a[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++) a[i] = (i * (u32)2654435761) + seed;
u32 sum = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)7000000; r++)
for (u32 i = (u32)0; i < (u32)N; i++) sum = sum + a[i];
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", sum, t1 - t0);
return 0;
}

Shifts and bitwise logic over an array.

Lines: xc 19, Objective-C 21, C++ 22, Swift 21.

// bit_ops — shifts and bitwise logic over an array.
// Exercises: shift lowering, and/or/xor, rotate idioms.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
u32 a[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++) a[i] = (i * (u32)2654435761) + seed;
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)600000; r++)
for (u32 i = (u32)0; i < (u32)N; i++)
acc = acc + (((a[i] + r) << (u32)3) | ((a[i] + r) >> (u32)5)) ^ (u32)0x0F0F0F0F;
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Data-dependent branches over an array.

Lines: xc 22, Objective-C 21, C++ 22, Swift 20.

// branch_mix — data-dependent branches over an array.
// Exercises: if-conversion, select formation, branch prediction behaviour.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
u32 a[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++) a[i] = (i * (u32)2654435761) + seed;
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)500000; r++)
for (u32 i = (u32)0; i < (u32)N; i++)
{
if ((a[i] & (u32)1) == (u32)0) acc = acc + a[i];
else acc = acc ^ a[i];
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

A small non-inlinable call chain in a hot loop.

Lines: xc 19, Objective-C 22, C++ 22, Swift 17.

// call_depth — a small non-inlinable call chain in a hot loop.
// Exercises: call overhead, leaf inlining, tail-call conversion.
#import "Stdio.xc"
#import "include/bench_time.xc"
u32 leaf(u32 x) { return (x * (u32)3) ^ (x >> (u32)2); }
u32 mid(u32 x) { return leaf(x) + leaf(x + (u32)1); }
u32 outer(u32 x) { return mid(x) ^ mid(x + (u32)2); }
i32 main(i32 argc, u8** argv)
{
u32 seed = (u32)argc;
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)2800000000; r++) acc = acc + outer(r + seed);
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Float multiply, accumulated in double.

Lines: xc 29, Objective-C 24, C++ 28, Swift 24.

// float_math — float multiply, accumulated in double.
// Exercises: float arithmetic, multiply-add fusion, float register pressure.
//
// Values are small exact integers and the accumulator is double, so the sum is
// exact whatever order the additions happen in. A vectorising compiler reorders
// them, and without exactness the two languages would disagree on rounding.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
float a[N]; float b[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++)
{ a[i] = (float)((i + seed) % (u32)16); b[i] = (float)((i % (u32)7) + (u32)1); }
double acc = 0.0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)400000; r++)
for (u32 i = (u32)0; i < (u32)N; i++)
acc = acc + (double)(a[i] * b[i]);
i64 t1 = bench_now_us();
// (u32)acc directly is UNDEFINED once the sum exceeds u32: at this
// iteration count acc reaches ~4.9e10, and the two compilers chose
// differently (0 against 4294967295). The double is still exact — well
// under 2^53 — so going through u64 and letting the u32 narrowing wrap,
// which IS defined, keeps the checksum meaningful and identical.
Stdio.printf("%lu %lld\n", (u32)((u64)acc), t1 - t0);
return 0;
}

An integer avalanche chain.

Lines: xc 19, Objective-C 22, C++ 20, Swift 17.

// hash_mix — an integer avalanche chain.
// Exercises: multiply, shift and xor chains with a serial dependency.
#import "Stdio.xc"
#import "include/bench_time.xc"
i32 main(i32 argc, u8** argv)
{
u32 h = (u32)argc;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)640000000; r++)
{
h = h ^ (h >> (u32)16);
h = h * (u32)2246822519;
h = h ^ (h >> (u32)13);
h = h + r;
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", h, t1 - t0);
return 0;
}

Integer add and xor over an array.

Lines: xc 29, Objective-C 30, C++ 24, Swift 18.

// int_accum — integer add and xor over an array.
// Exercises: loop induction, array addressing, u32 wraparound arithmetic.
//
// The array is seeded from argc so its contents are unknown at compile time,
// and the loop reads it every iteration. Without that both compilers reduce the
// whole loop to a closed form and the benchmark measures nothing.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
u32 a[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++)
a[i] = (i * (u32)2654435761) + seed;
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)600000; r++)
for (u32 i = (u32)0; i < (u32)N; i++)
acc = acc + (a[i] ^ acc);
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Integer multiply and divide by compile-time constants.

Lines: xc 19, Objective-C 21, C++ 21, Swift 18.

// int_muldiv — integer multiply and divide by compile-time constants.
// Exercises: strength reduction, magic-number division.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 1024
i32 main(i32 argc, u8** argv)
{
u32 a[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++) a[i] = (i * (u32)2654435761) + seed;
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)6800000; r++)
for (u32 i = (u32)0; i < (u32)N; i++)
acc = acc + ((a[i] * (u32)7) / (u32)3) + (a[i] / (u32)11);
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Multiply two small square matrices, repeatedly.

Lines: xc 27, Objective-C 29, C++ 29, Swift 30.

// matrix_mul — multiply two small square matrices, repeatedly.
// Exercises: triple-nested loops, strided addressing, accumulation.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define M 32
i32 main(i32 argc, u8** argv)
{
u32 a[M * M]; u32 b[M * M]; u32 c[M * M];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)(M * M); i++)
{ a[i] = (i + seed) & (u32)15; b[i] = (i ^ seed) & (u32)15; }
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)180000; r++)
for (u32 i = (u32)0; i < (u32)M; i++)
for (u32 j = (u32)0; j < (u32)M; j++)
{
u32 s = (u32)0;
for (u32 k = (u32)0; k < (u32)M; k++)
s = s + (a[i * (u32)M + k] * b[k * (u32)M + j]);
c[i * (u32)M + j] = s + r;
}
u32 acc = (u32)0;
for (u32 i = (u32)0; i < (u32)(M * M); i++) acc = acc + c[i];
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Multiply two float matrices, repeatedly.

Lines: xc 37, Objective-C 35, C++ 33, Swift 32.

// matrix_mul_f32 — multiply two float matrices, repeatedly.
// Exercises: a dense float matrix multiply (SME outer products on arm64 macOS
// with SME; NEON or SSE elsewhere). The values are small integers, so every
// product and sum is exact and the checksum is the same in every language,
// whether or not it fuses a*b + s. `:goal(speed)` lets the SME kernel skip
// its NaN check of C: only NaN payloads could differ, and there are no NaNs.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define M 128
i32 main(i32 argc, u8** argv)
{
float* a = new float[M * M];
float* b = new float[M * M];
float* c = new float[M * M];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)(M * M); i++)
{ a[i] = (float)((i + seed) & (u32)15); b[i] = (float)((i ^ seed) & (u32)15); }
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)10000; r++)
{
a[r % (u32)(M * M)] = (float)(r & (u32)15);
for (u32 i = (u32)0; i < (u32)M; i++) :goal(speed)
for (u32 j = (u32)0; j < (u32)M; j++)
{
float s = 0.0;
for (u32 k = (u32)0; k < (u32)M; k++)
s = s + a[i * (u32)M + k] * b[k * (u32)M + j];
c[i * (u32)M + j] = s;
}
acc = acc + (u32)c[(r * (u32)7919) % (u32)(M * M)];
}
for (u32 i = (u32)0; i < (u32)(M * M); i++) acc = acc + (u32)c[i];
i64 t1 = bench_now_us();
Stdio.printf("%u %lld\n", acc, t1 - t0);
return 0;
}

Copy between arrays element by element.

Lines: xc 21, Objective-C 24, C++ 24, Swift 20.

// mem_copy — copy between arrays element by element.
// Exercises: load/store pairing, the memcpy idiom, pointer induction.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 4096
i32 main(i32 argc, u8** argv)
{
u32 src[N]; u32 dst[N];
u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++) src[i] = i + seed;
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)6800000; r++)
{
for (u32 i = (u32)0; i < (u32)N; i++) dst[i] = src[i] + r;
acc = acc + dst[r % (u32)N];
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

A virtual method call per iteration.

Lines: xc 27, Objective-C 32, C++ 35, Swift 24.

// method_call — a virtual method call per iteration.
// Exercises: dispatch cost, inlining across a method boundary.
// The class is chosen at run time and score() takes the iteration number, so
// the call can be neither resolved at compile time nor hoisted out of the loop.
#import "Stdio.xc"
#import "include/bench_time.xc"
class Shape { u32 k; void init(u32 x) { k = x; } u32 score(u32 x) { return k ^ x; } }
class Boxy : Shape
{
void init(u32 x) { super.init(x); }
u32 score(u32 x) { return (k ^ x) + (u32)1; }
}
i32 main(i32 argc, u8** argv)
{
u32 seed = (u32)argc;
Shape* s = new Shape(seed);
if ((argc & 1) == 0)
s = new Boxy(seed);
u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)1600000000; r++) acc = acc + s.score(r);
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Dispatch through a mixed array of subclasses.

Lines: xc 25, Objective-C 35, C++ 52, Swift 40.

// poly_dispatch — dispatch through a mixed array of subclasses.
// Exercises: virtual dispatch the compiler cannot devirtualise.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 256
class Op { u32 k; void init(u32 x) { k = x; } u32 apply(u32 v) { return v + k; } }
class OpMul : Op { void init(u32 x) { super.init(x); } u32 apply(u32 v) { return v * (k | (u32)1); } }
class OpXor : Op { void init(u32 x) { super.init(x); } u32 apply(u32 v) { return v ^ k; } }
i32 main(i32 argc, u8** argv)
{
Op* ops[N]; u32 seed = (u32)argc;
for (u32 i = (u32)0; i < (u32)N; i++)
{
if ((i % (u32)3) == (u32)0) ops[i] = new Op(i + seed);
else if ((i % (u32)3) == (u32)1) ops[i] = new OpMul(i + seed);
else ops[i] = new OpXor(i + seed);
}
u32 acc = (u32)1;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)3000000; r++)
for (u32 i = (u32)0; i < (u32)N; i++) acc = ops[i].apply(acc) + r;
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Sieve of Eratosthenes over a fixed range, repeatedly.

Lines: xc 25, Objective-C 28, C++ 30, Swift 27.

// sieve — sieve of Eratosthenes over a fixed range, repeatedly.
// Exercises: byte array writes, strided inner loops, division-free stepping.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 8192
i32 main(i32 argc, u8** argv)
{
u8 flags[N]; u32 seed = (u32)argc; u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)150000; r++)
{
for (u32 i = (u32)0; i < (u32)N; i++) flags[i] = (u8)1;
u32 count = (u32)0;
for (u32 i = (u32)2; i < (u32)N; i++)
if (flags[i] != (u8)0)
{
count = count + (u32)1;
for (u32 j = i + i; j < (u32)N; j = j + i) flags[j] = (u8)0;
}
acc = acc + count + (r & seed);
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Insertion sort of a small array, repeatedly.

Lines: xc 25, Objective-C 28, C++ 30, Swift 29.

// sort_small — insertion sort of a small array, repeatedly.
// Exercises: nested loops, data-dependent branches, swaps.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 64
i32 main(i32 argc, u8** argv)
{
u32 a[N]; u32 seed = (u32)argc; u32 acc = (u32)0;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)3200000; r++)
{
for (u32 i = (u32)0; i < (u32)N; i++)
a[i] = ((i * (u32)2654435761) ^ (r * (u32)40503)) + seed;
for (u32 i = (u32)1; i < (u32)N; i++)
{
u32 v = a[i]; u32 j = i;
while (j > (u32)0 && a[j - (u32)1] > v) { a[j] = a[j - (u32)1]; j = j - (u32)1; }
a[j] = v;
}
acc = acc + a[0] + a[N - 1];
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Scan bytes for a delimiter and checksum them.

Lines: xc 22, Objective-C 24, C++ 26, Swift 23.

// string_scan — scan bytes for a delimiter and checksum them.
// Exercises: byte loads, comparisons, narrow-type arithmetic.
#import "Stdio.xc"
#import "include/bench_time.xc"
#define N 8192
i32 main(i32 argc, u8** argv)
{
u8 buf[N]; u32 seed = (u32)argc; u32 acc = (u32)0;
for (u32 i = (u32)0; i < (u32)N; i++)
buf[i] = (u8)(((i * (u32)31) + seed) & (u32)127);
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)3800000; r++)
{
u32 n = (u32)0;
for (u32 i = (u32)0; i < (u32)N; i++)
{ if (buf[i] == (u8)44) n = n + (u32)1; acc = acc + (u32)buf[i]; }
acc = acc + n;
}
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

Pass and return a small struct by value.

Lines: xc 17, Objective-C 19, C++ 23, Swift 26.

// struct_copy — pass and return a small struct by value.
// Exercises: struct copy semantics, field layout, argument passing.
#import "Stdio.xc"
#import "include/bench_time.xc"
struct Pt { u32 x, y; }
Pt bump(Pt p, u32 d) { Pt q; q.x = p.x + d; q.y = p.y ^ d; return q; }
i32 main(i32 argc, u8** argv)
{
u32 seed = (u32)argc; u32 acc = (u32)0;
Pt p; p.x = seed; p.y = seed;
i64 t0 = bench_now_us();
for (u32 r = (u32)0; r < (u32)2200000000; r++)
{ p = bump(p, r); acc = acc + p.x + p.y; }
i64 t1 = bench_now_us();
Stdio.printf("%lu %lld\n", acc, t1 - t0);
return 0;
}

The programs behind the performance page’s GPU tables. Each is one xc program: its work is a par block, a loop whose iterations are independent, and the runtime runs it on the CPU’s threads or on the GPU (Metal on a Mac, CUDA or Vulkan on Windows, Vulkan on Linux and Android, WebGPU in the browser), whichever it finds faster. Nothing in the source names a device, a kernel language or a thread.

Escape-time iteration counts over a 2048x2048 grid, one item

Lines: 51.

// mandelbrot — escape-time iteration counts over a 2048x2048 grid, one item
// per pixel: arithmetic only, with a data-dependent loop. Prints
// "<checksum> <best_us> <first_us>"; the checksum is the total iteration count to
// the nearest million, coarse enough that fast GPU maths cannot move it.
#import "Stdio.xc"
#import "Par.xc"
#import "../src/include/bench_time.xc"
#define W 2048
#define SIZE (2048 * 2048)
i32 main(void)
{
// Eight runs: the first carries one-off costs (a GPU builds its kernel
// then), so the figure is the best run; the first is printed too. Auto
// decides after its fourth (two on each device), and uses the rest.
u32 total = (u32)0;
i64 best = (i64)0;
i64 first = (i64)0;
for (u32 rep in 0..8)
{
total = (u32)0;
i64 t0 = bench_now_us();
par mandel :reduce(+ total)
{
for (u32 i in 0..SIZE)
{
float cx = (float)(i % (u32)W) * (3.0f / 2048.0f) - 2.0f;
float cy = (float)(i / (u32)W) * (3.0f / 2048.0f) - 1.5f;
float x = 0.0f;
float y = 0.0f;
u32 k = (u32)0;
while (k < (u32)256 && x * x + y * y < 4.0f)
{
float xt = x * x - y * y + cx;
y = 2.0f * x * y + cy;
x = xt;
k = k + (u32)1;
}
total = total + k;
}
}
i64 t1 = bench_now_us();
if (rep == (u32)0)
first = t1 - t0;
if (rep == (u32)0 || t1 - t0 < best)
best = t1 - t0;
}
Stdio.printf("%u %lld %lld\n", (total + (u32)500000) / (u32)1000000, best, first);
return 0;
}

The gravitational force on each of 8192 bodies from all the others:

Lines: 63.

// nbody — the gravitational force on each of 8192 bodies from all the others:
// O(n^2) arithmetic over small arrays, one item per body. Prints
// "<checksum> <best_us> <first_us>"; the checksum counts the bodies pulled right
// (positive x force), a coarse figure fast GPU maths does not move.
#import "Stdio.xc"
#import "Math.xc"
#import "Par.xc"
#import "../src/include/bench_time.xc"
#define N 8192
float px[N];
float py[N];
float mass[N];
float fx[N];
i32 main(void)
{
u32 seed = (u32)7;
for (u32 i in 0..N)
{
seed = seed * (u32)1664525 + (u32)1013904223;
px[i] = (float)(seed >> (u32)8) / 16777216.0f;
seed = seed * (u32)1664525 + (u32)1013904223;
py[i] = (float)(seed >> (u32)8) / 16777216.0f;
mass[i] = 1.0f + (float)(i % (u32)7);
}
// Eight runs: the first carries one-off costs (a GPU builds its kernel
// then), so the figure is the best run; the first is printed too. Auto
// decides after its fourth (two on each device), and uses the rest.
u32 right = (u32)0;
i64 best = (i64)0;
i64 first = (i64)0;
for (u32 rep in 0..8)
{
right = (u32)0;
i64 t0 = bench_now_us();
par forces :reduce(+ right)
{
for (u32 i in 0..N)
{
float ax = 0.0f;
for (u32 j in 0..N)
{
float dx = px[j] - px[i];
float dy = py[j] - py[i];
float d2 = dx * dx + dy * dy + 0.0001f;
ax = ax + mass[j] * dx / (d2 * Math.sqrt(d2));
}
fx[i] = ax;
if (ax > 0.0f)
right = right + (u32)1;
}
}
i64 t1 = bench_now_us();
if (rep == (u32)0)
first = t1 - t0;
if (rep == (u32)0 || t1 - t0 < best)
best = t1 - t0;
}
Stdio.printf("%u %lld %lld\n", (right + (u32)50) / (u32)100, best, first);
return 0;
}

Ken Perlin’s improved noise, 2D, four octaves, over a 2048x2048

Lines: 102.

// perlin — Ken Perlin's improved noise, 2D, four octaves, over a 2048x2048
// image: a permutation table read at data-dependent places, helper calls,
// and one store per pixel. Prints "<checksum> <best_us> <first_us>"; the checksum is
// the mean grey level to a tenth.
#import "Stdio.xc"
#import "Par.xc"
#import "../src/include/bench_time.xc"
#define SIZE (2048 * 2048)
u32 perm[512];
u8 img[SIZE];
float fade(float t) { return t * t * t * (t * (t * 6.0f - 15.0f) + 10.0f); }
float lerp(float t, float a, float b) { return a + t * (b - a); }
float grad(u32 h, float x, float y)
{
u32 g = h & (u32)7;
float u = g < (u32)4 ? x : y;
float v = g < (u32)4 ? y : x;
float a = (g & (u32)1) != (u32)0 ? 0.0f - u : u;
float b = (g & (u32)2) != (u32)0 ? 0.0f - 2.0f * v : 2.0f * v;
return a + b;
}
i32 main(void)
{
u32 seed = (u32)12345;
for (u32 i in 0..256)
perm[i] = i;
for (u32 i in 0..256)
{
seed = seed * (u32)1103515245 + (u32)12345;
u32 j = i + (seed >> (u32)16) % ((u32)256 - i);
u32 t = perm[i];
perm[i] = perm[j];
perm[j] = t;
}
for (u32 i in 0..256)
perm[i + (u32)256] = perm[i];
// Eight runs: the first carries one-off costs (a GPU builds its kernel
// then), so the figure is the best run; the first is printed too. Auto
// decides after its fourth (two on each device), and uses the rest.
u32 total = (u32)0;
i64 best = (i64)0;
i64 first = (i64)0;
for (u32 rep in 0..8)
{
total = (u32)0;
i64 t0 = bench_now_us();
par perlin :reduce(+ total)
{
for (u32 i in 0..SIZE)
{
float x = (float)(i % (u32)2048) / 256.0f;
float y = (float)(i / (u32)2048) / 256.0f;
float sum = 0.0f;
float amp = 1.0f;
float norm = 0.0f;
for (u32 o in 0..4)
{
u32 xi = (u32)x;
u32 yi = (u32)y;
float xf = x - (float)xi;
float yf = y - (float)yi;
u32 X = xi & (u32)255;
u32 Y = yi & (u32)255;
u32 aa = perm[perm[X] + Y];
u32 ab = perm[perm[X] + Y + (u32)1];
u32 ba = perm[perm[X + (u32)1] + Y];
u32 bb = perm[perm[X + (u32)1] + Y + (u32)1];
float u = fade(xf);
float v = fade(yf);
float n = lerp(v, lerp(u, grad(aa, xf, yf), grad(ba, xf - 1.0f, yf)),
lerp(u, grad(ab, xf, yf - 1.0f), grad(bb, xf - 1.0f, yf - 1.0f)));
sum = sum + n * amp;
norm = norm + amp;
amp = amp * 0.5f;
x = x * 2.0f;
y = y * 2.0f;
}
float s = (sum / norm) * 0.5f + 0.5f;
if (s < 0.0f)
s = 0.0f;
if (s > 1.0f)
s = 1.0f;
u32 p = (u32)(s * 255.0f);
img[i] = (u8)p;
total = total + p;
}
}
i64 t1 = bench_now_us();
if (rep == (u32)0)
first = t1 - t0;
if (rep == (u32)0 || t1 - t0 < best)
best = t1 - t0;
}
u32 tenths = (total * (u32)10 + (u32)(SIZE / 2)) / (u32)SIZE;
Stdio.printf("%u %lld %lld\n", tenths, best, first);
return 0;
}

Y = a*x + y over 16M integers: one multiply-add per element and

Lines: 48.

// saxpy — y = a*x + y over 16M integers: one multiply-add per element and
// two arrays to move, so memory, not arithmetic, sets the pace. On a GPU the
// copies in and out cost more than the work: this is the block auto keeps on
// the CPU. Prints "<checksum> <best_us> <first_us>"; the checksum is exact.
#import "Stdio.xc"
#import "Par.xc"
#import "../src/include/bench_time.xc"
#define SIZE (16 * 1024 * 1024)
u32 xs[SIZE];
u32 ys[SIZE];
i32 main(void)
{
for (u32 i in 0..SIZE)
{
xs[i] = i * (u32)2654435761;
ys[i] = i;
}
// Eight runs: the first carries one-off costs (a GPU builds its kernel
// then), so the figure is the best run; the first is printed too. Auto
// decides after its fourth (two on each device), and uses the rest.
u32 total = (u32)0;
i64 best = (i64)0;
i64 first = (i64)0;
for (u32 rep in 0..8)
{
total = (u32)0;
i64 t0 = bench_now_us();
par saxpy :reduce(+ total)
{
for (u32 i in 0..SIZE)
{
u32 v = (u32)3 * xs[i] + ys[i];
ys[i] = v;
total = total + v;
}
}
i64 t1 = bench_now_us();
if (rep == (u32)0)
first = t1 - t0;
if (rep == (u32)0 || t1 - t0 < best)
best = t1 - t0;
}
Stdio.printf("%u %lld %lld\n", total, best, first);
return 0;
}