Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions build.rs
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,19 @@ fn main() {
// with a clang that doesn't understand preserve_none, the
// build fails loudly instead of producing a subtly broken
// interpreter.
// Round the multiply before the add, the way Cranelift and
// LLVM do. Left to clang's default, the fused arithmetic
// handlers (op_f32x4_muladd_set, op_fused_get_get_fmul_fadd_f,
// ...) contract into fmla/fmadd, which rounds once instead of
// twice — so the stack VM returns different numbers than the
// JIT for the same source, and the stack VM is the shipped iOS
// backend (see src/ffi.rs). Worse, whether it happens at all
// depends on the host compiler's default, so the same source
// can behave differently depending on who built it. Backend
// agreement is worth more here than the last ulp; the cost is
// ~6% on f32x4 multiply-accumulate and nothing measurable
// elsewhere. tests/cases/simd/f32x4_fma_rounding.lyte guards it.
.flag("-ffp-contract=off")
.flag("-Werror=unknown-attributes")
.compile("stack_interp");
println!("cargo:rustc-cfg=has_stack_interp");
Expand Down
339 changes: 278 additions & 61 deletions src/stack_codegen.rs

Large diffs are not rendered by default.

30 changes: 30 additions & 0 deletions src/stack_depth.rs
Original file line number Diff line number Diff line change
Expand Up @@ -400,6 +400,32 @@ pub fn stack_delta(op: &StackOp) -> i32 {
| StackOp::FusedGetSet8D(_)
| StackOp::FusedF64ConstDGtJumpIfZeroD(_, _)
| StackOp::FusedGetF64ConstDGtJumpIfZeroD(_, _, _) => 0,

// f32x4: the plain forms pop their operand addresses and push the
// destination slot's address; the constructors take their lanes
// from the float window, so they only push. The `*Store` forms
// pop a destination address too and push nothing.
StackOp::F32x4Add(_)
| StackOp::F32x4Sub(_)
| StackOp::F32x4Mul(_)
| StackOp::F32x4Div(_) => -1,
StackOp::F32x4Neg(_) => 0,
StackOp::F32x4Build(_) | StackOp::F32x4Splat(_) => 1,
StackOp::F32x4AddStore
| StackOp::F32x4SubStore
| StackOp::F32x4MulStore
| StackOp::F32x4DivStore => -3,
StackOp::F32x4NegStore => -2,
StackOp::F32x4BuildStore | StackOp::F32x4SplatStore => -1,

// Three-address forms read and write frame slots only.
StackOp::F32x4Add3(_, _, _)
| StackOp::F32x4Sub3(_, _, _)
| StackOp::F32x4Mul3(_, _, _)
| StackOp::F32x4Div3(_, _, _)
| StackOp::F32x4Neg2(_, _)
| StackOp::F32x4MulAddSet(_, _, _, _)
| StackOp::F32x4MulSubSet(_, _, _, _) => 0,
}
}

Expand Down Expand Up @@ -622,6 +648,10 @@ pub fn float_stack_delta(op: &StackOp) -> i32 {
// f-window loads push f0
StackOp::LoadF32F | StackOp::LoadF32OffF(_) => 1,

// f32x4 constructors take their lanes off the float window.
StackOp::F32x4Build(_) | StackOp::F32x4BuildStore => -4,
StackOp::F32x4Splat(_) | StackOp::F32x4SplatStore => -1,

_ => 0,
}
}
221 changes: 221 additions & 0 deletions src/stack_interp.c
Original file line number Diff line number Diff line change
Expand Up @@ -113,6 +113,22 @@ static inline void store_f64_unaligned(void* p, double v) {
memcpy(p, &v, sizeof(v));
}

// f32x4 lives in 16 bytes of frame memory. alloc_memory rounds frame
// allocations to 8-byte slots, so those 16 bytes are only 8-byte aligned:
// go through memcpy so the compiler emits an unaligned vector load/store
// (`ldur q` on aarch64, `movups` on x86-64) rather than assuming 16.
typedef float v4f __attribute__((vector_size(16)));

static inline v4f load_v4f(const void* p) {
v4f v;
memcpy(&v, p, sizeof(v));
return v;
}

static inline void store_v4f(void* p, v4f v) {
memcpy(p, &v, sizeof(v));
}

// ============================================================================
// Integer power
// ============================================================================
Expand Down Expand Up @@ -219,6 +235,13 @@ static int64_t ipow(int64_t base, uint32_t exp) {
} while(0)

#define FDROP1() do { f0 = f1; f1 = f2; f2 = f3; f3 = *--fsp; } while(0)

// Drop 4 floats (the whole window) — the f32x4 constructors consume all
// four lanes at once, so the window refills entirely from the spill area.
#define FDROP4() do { \
f0 = *(fsp - 1); f1 = *(fsp - 2); f2 = *(fsp - 3); f3 = *(fsp - 4); \
fsp -= 4; \
} while(0)
#define FBINOP_SHIFT() do { f1 = f2; f2 = f3; f3 = *--fsp; } while(0)

// f64 TOS window push/pop — exact mirror of the f32 window above, but
Expand Down Expand Up @@ -2318,6 +2341,204 @@ HANDLER(op_fused_get_f64const_dgt_jiz_d) {
NEXT();
}

// ============================================================================
// f32x4 SIMD
// ============================================================================
//
// An f32x4 travels as an address in the int window, like every other
// pointer-represented type. Each op loads whole 16-byte vectors, does the
// arithmetic on a `v4f`, and stores the result — one SIMD instruction per
// operation instead of the four per-lane load/op/store trips through the
// float window that the scalarized path emitted.
//
// The plain forms name their destination frame slot in imm[0] and push its
// address (expression position); the `*Store` forms take the destination
// address off the top of the int window and push nothing, so an assignment
// writes straight into its destination with no temp and no 16-byte copy.
// The operands are read into registers before the store, so a destination
// aliasing either operand (`v = v * k`) is fine.

HANDLER(op_f32x4_add) {
v4f a = load_v4f((const void*)t1);
v4f b = load_v4f((const void*)t0);
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, a + b);
t0 = (uint64_t)dst;
BINOP_SHIFT();
NEXT();
}

HANDLER(op_f32x4_sub) {
v4f a = load_v4f((const void*)t1);
v4f b = load_v4f((const void*)t0);
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, a - b);
t0 = (uint64_t)dst;
BINOP_SHIFT();
NEXT();
}

HANDLER(op_f32x4_mul) {
v4f a = load_v4f((const void*)t1);
v4f b = load_v4f((const void*)t0);
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, a * b);
t0 = (uint64_t)dst;
BINOP_SHIFT();
NEXT();
}

HANDLER(op_f32x4_div) {
v4f a = load_v4f((const void*)t1);
v4f b = load_v4f((const void*)t0);
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, a / b);
t0 = (uint64_t)dst;
BINOP_SHIFT();
NEXT();
}

HANDLER(op_f32x4_neg) {
v4f a = load_v4f((const void*)t0);
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, -a);
t0 = (uint64_t)dst;
NEXT();
}

// Lanes are pushed 0,1,2,3, so f0 holds lane 3 and f3 holds lane 0.
HANDLER(op_f32x4_build) {
v4f v = { f3, f2, f1, f0 };
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, v);
FDROP4();
PUSH((uint64_t)dst);
NEXT();
}

HANDLER(op_f32x4_splat) {
v4f v = { f0, f0, f0, f0 };
void* dst = (void*)(locals + pc->imm[0]);
store_v4f(dst, v);
FDROP1();
PUSH((uint64_t)dst);
NEXT();
}

// Store forms: t0 = destination address, t1 = b, t2 = a.
HANDLER(op_f32x4_add_store) {
v4f a = load_v4f((const void*)t2);
v4f b = load_v4f((const void*)t1);
store_v4f((void*)t0, a + b);
DROP3();
NEXT();
}

HANDLER(op_f32x4_sub_store) {
v4f a = load_v4f((const void*)t2);
v4f b = load_v4f((const void*)t1);
store_v4f((void*)t0, a - b);
DROP3();
NEXT();
}

HANDLER(op_f32x4_mul_store) {
v4f a = load_v4f((const void*)t2);
v4f b = load_v4f((const void*)t1);
store_v4f((void*)t0, a * b);
DROP3();
NEXT();
}

HANDLER(op_f32x4_div_store) {
v4f a = load_v4f((const void*)t2);
v4f b = load_v4f((const void*)t1);
store_v4f((void*)t0, a / b);
DROP3();
NEXT();
}

HANDLER(op_f32x4_neg_store) {
v4f a = load_v4f((const void*)t1);
store_v4f((void*)t0, -a);
DROP2();
NEXT();
}

HANDLER(op_f32x4_build_store) {
v4f v = { f3, f2, f1, f0 };
store_v4f((void*)t0, v);
FDROP4();
DROP1();
NEXT();
}

HANDLER(op_f32x4_splat_store) {
v4f v = { f0, f0, f0, f0 };
store_v4f((void*)t0, v);
FDROP1();
DROP1();
NEXT();
}

// Three-address forms: every operand and the destination is a frame slot,
// so nothing touches the operand stack. imm[2] of the multiply-accumulate
// ops packs c in the low half and dst in the high half.
//
// `a * b + c` contracts to a single fmla.4s. That makes these ops round
// once where a separate multiply and add would round twice — the same
// trade the scalar op_fused_get_get_fmul_fadd_f already makes.

HANDLER(op_f32x4_add3) {
v4f a = load_v4f(locals + pc->imm[0]);
v4f b = load_v4f(locals + pc->imm[1]);
store_v4f(locals + pc->imm[2], a + b);
NEXT();
}

HANDLER(op_f32x4_sub3) {
v4f a = load_v4f(locals + pc->imm[0]);
v4f b = load_v4f(locals + pc->imm[1]);
store_v4f(locals + pc->imm[2], a - b);
NEXT();
}

HANDLER(op_f32x4_mul3) {
v4f a = load_v4f(locals + pc->imm[0]);
v4f b = load_v4f(locals + pc->imm[1]);
store_v4f(locals + pc->imm[2], a * b);
NEXT();
}

HANDLER(op_f32x4_div3) {
v4f a = load_v4f(locals + pc->imm[0]);
v4f b = load_v4f(locals + pc->imm[1]);
store_v4f(locals + pc->imm[2], a / b);
NEXT();
}

HANDLER(op_f32x4_neg2) {
v4f a = load_v4f(locals + pc->imm[0]);
store_v4f(locals + pc->imm[1], -a);
NEXT();
}

HANDLER(op_f32x4_muladd_set) {
v4f a = load_v4f(locals + pc->imm[0]);
v4f b = load_v4f(locals + pc->imm[1]);
v4f c = load_v4f(locals + (pc->imm[2] & 0xFFFFu));
store_v4f(locals + (pc->imm[2] >> 16), a * b + c);
NEXT();
}

HANDLER(op_f32x4_mulsub_set) {
v4f a = load_v4f(locals + pc->imm[0]);
v4f b = load_v4f(locals + pc->imm[1]);
v4f c = load_v4f(locals + (pc->imm[2] & 0xFFFFu));
store_v4f(locals + (pc->imm[2] >> 16), a * b - c);
NEXT();
}

// ============================================================================
// Entry point
// ============================================================================
Expand Down
62 changes: 62 additions & 0 deletions src/stack_interp_bridge.rs
Original file line number Diff line number Diff line change
Expand Up @@ -364,6 +364,27 @@ extern "C" {
fn op_fused_get_set8_d();
fn op_fused_f64const_dgt_jiz_d();
fn op_fused_get_f64const_dgt_jiz_d();
fn op_f32x4_add();
fn op_f32x4_sub();
fn op_f32x4_mul();
fn op_f32x4_div();
fn op_f32x4_neg();
fn op_f32x4_build();
fn op_f32x4_splat();
fn op_f32x4_add_store();
fn op_f32x4_sub_store();
fn op_f32x4_mul_store();
fn op_f32x4_div_store();
fn op_f32x4_neg_store();
fn op_f32x4_build_store();
fn op_f32x4_splat_store();
fn op_f32x4_add3();
fn op_f32x4_sub3();
fn op_f32x4_mul3();
fn op_f32x4_div3();
fn op_f32x4_neg2();
fn op_f32x4_muladd_set();
fn op_f32x4_mulsub_set();
}

/// Get the C handler function pointer for a StackOp.
Expand Down Expand Up @@ -678,6 +699,29 @@ fn handler_for(op: &StackOp) -> *const () {
StackOp::FusedGetF64ConstDGtJumpIfZeroD(_, _, _) => {
op_fused_get_f64const_dgt_jiz_d as *const ()
}

// === f32x4 SIMD ops ===
StackOp::F32x4Add(_) => op_f32x4_add as *const (),
StackOp::F32x4Sub(_) => op_f32x4_sub as *const (),
StackOp::F32x4Mul(_) => op_f32x4_mul as *const (),
StackOp::F32x4Div(_) => op_f32x4_div as *const (),
StackOp::F32x4Neg(_) => op_f32x4_neg as *const (),
StackOp::F32x4Build(_) => op_f32x4_build as *const (),
StackOp::F32x4Splat(_) => op_f32x4_splat as *const (),
StackOp::F32x4AddStore => op_f32x4_add_store as *const (),
StackOp::F32x4SubStore => op_f32x4_sub_store as *const (),
StackOp::F32x4MulStore => op_f32x4_mul_store as *const (),
StackOp::F32x4DivStore => op_f32x4_div_store as *const (),
StackOp::F32x4NegStore => op_f32x4_neg_store as *const (),
StackOp::F32x4BuildStore => op_f32x4_build_store as *const (),
StackOp::F32x4SplatStore => op_f32x4_splat_store as *const (),
StackOp::F32x4Add3(_, _, _) => op_f32x4_add3 as *const (),
StackOp::F32x4Sub3(_, _, _) => op_f32x4_sub3 as *const (),
StackOp::F32x4Mul3(_, _, _) => op_f32x4_mul3 as *const (),
StackOp::F32x4Div3(_, _, _) => op_f32x4_div3 as *const (),
StackOp::F32x4Neg2(_, _) => op_f32x4_neg2 as *const (),
StackOp::F32x4MulAddSet(_, _, _, _) => op_f32x4_muladd_set as *const (),
StackOp::F32x4MulSubSet(_, _, _, _) => op_f32x4_mulsub_set as *const (),
}
}

Expand Down Expand Up @@ -911,6 +955,24 @@ fn encode_imm(op: &StackOp, func_idx: u32) -> [u64; 3] {
StackOp::FusedGetF64ConstDGtJumpIfZeroD(n, v, off) => {
[(*n as u64) * 8, f64::to_bits(*v), *off as i64 as u64]
}
// f32x4 ops name their destination frame slot.
StackOp::F32x4Add(d)
| StackOp::F32x4Sub(d)
| StackOp::F32x4Mul(d)
| StackOp::F32x4Div(d)
| StackOp::F32x4Neg(d)
| StackOp::F32x4Build(d)
| StackOp::F32x4Splat(d) => [*d as u64, 0, 0],
StackOp::F32x4Add3(a, b, d)
| StackOp::F32x4Sub3(a, b, d)
| StackOp::F32x4Mul3(a, b, d)
| StackOp::F32x4Div3(a, b, d) => [*a as u64, *b as u64, *d as u64],
StackOp::F32x4Neg2(a, d) => [*a as u64, *d as u64, 0],
// imm[2] packs c in the low half, dst in the high half.
StackOp::F32x4MulAddSet(a, b, c, d) | StackOp::F32x4MulSubSet(a, b, c, d) => {
[*a as u64, *b as u64, (*c as u64) | ((*d as u64) << 16)]
}

_ => [0, 0, 0],
}
}
Expand Down
Loading
Loading