From cba402a051f029f6932b8f3b2150748e2e10a33f Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 23 Sep 2026 20:07:00 +0000 Subject: [PATCH] VM: trap on 24-bit integer overflow instead of wrapping INT_ADD/INT_SUB/INT_MUL now return VmError::IntOverflow { pc } when the exact result falls outside [-2^23, 2^23 - 1]. The constant folder leaves out-of-range arithmetic unfolded so the VM traps at runtime, and the emitter rejects integer literals outside the range instead of truncating them. Documented in VM.md. Closes #16 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01L5Xv4Npfn2U5QBKUqV9Dbu --- VM.md | 8 +- crates/encore_compiler/src/pass/asm_emit.rs | 5 ++ .../cps_simplify/simpl_03_constant_fold.rs | 20 +++-- crates/encore_compiler/tests/emit_tests.rs | 28 ++++++ .../encore_compiler/tests/simplify_tests.rs | 44 +++++++++ crates/encore_vm/src/error.rs | 4 + crates/encore_vm/src/value.rs | 10 +++ crates/encore_vm/src/vm.rs | 17 +++- crates/encore_vm/tests/vm_tests.rs | 90 +++++++++++++++++++ 9 files changed, 211 insertions(+), 15 deletions(-) diff --git a/VM.md b/VM.md index aac2588..c2d14ed 100644 --- a/VM.md +++ b/VM.md @@ -23,7 +23,7 @@ Every runtime value is a **packed 32-bit word**: **Functions** (`TYP_FUNC`) are bare function values with no captures. The code pointer is stored inline — no heap allocation needed. **Closures** (`TYP_CLOS`) always point to a heap object whose header carries `env_len` (capture count) and `code_ptr`. -**Integers** use the upper 24 bits as a signed value (approx. ±8M range). `int_value()` recovers the `i32` via arithmetic right shift. +**Integers** use the upper 24 bits as a signed value, range `[-2^23, 2^23 - 1]` (`INT_MIN`/`INT_MAX` in `value.rs`). `int_value()` recovers the `i32` via arithmetic right shift. Arithmetic does not wrap: `INT_ADD`, `INT_SUB` and `INT_MUL` return `VmError::IntOverflow { pc }` when the exact result falls outside this range. The compiler keeps the same contract: the constant folder leaves out-of-range arithmetic unfolded (so it traps at runtime), and the emitter rejects integer literals outside the range. **`HeapAddress::NULL`** (`0xFFFF`) marks nullary constructors that have no heap allocation. @@ -126,9 +126,9 @@ Arguments `A1`–`A8` are staged by the compiler via `MOV` instructions before ` | `INT_0` | `18` | `rd: Reg` | `regs[rd] = Value::int(0)` | | `INT_1` | `19` | `rd: Reg` | `regs[rd] = Value::int(1)` | | `INT_2` | `1A` | `rd: Reg` | `regs[rd] = Value::int(2)` | -| `INT_ADD` | `11` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = int(regs[ra] + regs[rb])` (wrapping) | -| `INT_SUB` | `12` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = int(regs[ra] - regs[rb])` (wrapping) | -| `INT_MUL` | `13` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = int(regs[ra] * regs[rb])` (wrapping) | +| `INT_ADD` | `11` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = int(regs[ra] + regs[rb])`; `IntOverflow` error if the result is out of 24-bit range | +| `INT_SUB` | `12` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = int(regs[ra] - regs[rb])`; `IntOverflow` error if the result is out of 24-bit range | +| `INT_MUL` | `13` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = int(regs[ra] * regs[rb])`; `IntOverflow` error if the result is out of 24-bit range | | `INT_EQ` | `14` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = ctor(1, NULL)` if equal, `ctor(0, NULL)` otherwise | | `INT_LT` | `15` | `rd: Reg`, `ra: Reg`, `rb: Reg` | `regs[rd] = ctor(1, NULL)` if `a < b`, `ctor(0, NULL)` otherwise | | `INT_BYTE` | `16` | `rd: Reg`, `rs: Reg` | Convert integer 0–255 to a single-byte `Bytes` value; error if out of range | diff --git a/crates/encore_compiler/src/pass/asm_emit.rs b/crates/encore_compiler/src/pass/asm_emit.rs index 4cb7691..e1c1691 100644 --- a/crates/encore_compiler/src/pass/asm_emit.rs +++ b/crates/encore_compiler/src/pass/asm_emit.rs @@ -1,4 +1,5 @@ use encore_vm::opcode; +use encore_vm::value::{int_in_range, INT_MAX, INT_MIN}; use crate::ir::asm::{ContLam, Expr, Fun, Module, Reg, Val}; use crate::ir::prim::{PrimOp, IntOp, BytesOp}; @@ -162,6 +163,10 @@ impl<'a> Emitter<'a> { self.emit_u8(dest); } _ => { + assert!( + int_in_range(*n), + "integer literal {n} out of 24-bit range [{INT_MIN}, {INT_MAX}]", + ); self.emit_u8(opcode::INT); self.emit_u8(dest); let bits = *n as u32; diff --git a/crates/encore_compiler/src/pass/cps_simplify/simpl_03_constant_fold.rs b/crates/encore_compiler/src/pass/cps_simplify/simpl_03_constant_fold.rs index 4d6a0e7..e84bf52 100644 --- a/crates/encore_compiler/src/pass/cps_simplify/simpl_03_constant_fold.rs +++ b/crates/encore_compiler/src/pass/cps_simplify/simpl_03_constant_fold.rs @@ -12,9 +12,14 @@ // // let s = bytes [68 69] in bytes_len s ──► 2 // +// Integer arithmetic whose exact result falls outside the VM's 24-bit +// range is left unfolded, so the VM traps with `IntOverflow` at runtime. +// use std::collections::HashMap; +use encore_vm::value::int_in_range; + use crate::ir::cps::{Case, Cont, Expr, Fun, Tag, Val}; use crate::ir::cps_traversal::CPSTransformer; use crate::ir::prim::{BytesOp, IntOp, PrimOp}; @@ -162,7 +167,7 @@ fn try_fold_prim(op: PrimOp, args: &[String], env: &Env) -> Option { PrimOp::Int(op) => { let a = get_int(env, &args[0])?; let b = get_int(env, &args[1])?; - Some(eval_int_binop(op, a, b)) + eval_int_binop(op, a, b) } PrimOp::Bytes(BytesOp::Len) => { let bs = get_bytes(env, &args[0])?; @@ -195,13 +200,14 @@ fn try_fold_prim(op: PrimOp, args: &[String], env: &Env) -> Option { } } -fn eval_int_binop(op: IntOp, a: i32, b: i32) -> Val { +fn eval_int_binop(op: IntOp, a: i32, b: i32) -> Option { + let arith = |n: Option| n.filter(|&n| int_in_range(n)).map(Val::Int); match op { - IntOp::Add => Val::Int(a.wrapping_add(b)), - IntOp::Sub => Val::Int(a.wrapping_sub(b)), - IntOp::Mul => Val::Int(a.wrapping_mul(b)), - IntOp::Eq => if a == b { Val::TRUE } else { Val::FALSE }, - IntOp::Lt => if a < b { Val::TRUE } else { Val::FALSE }, + IntOp::Add => arith(a.checked_add(b)), + IntOp::Sub => arith(a.checked_sub(b)), + IntOp::Mul => arith(a.checked_mul(b)), + IntOp::Eq => Some(if a == b { Val::TRUE } else { Val::FALSE }), + IntOp::Lt => Some(if a < b { Val::TRUE } else { Val::FALSE }), IntOp::Byte => unreachable!("Byte is unary and handled by try_fold_prim"), } } diff --git a/crates/encore_compiler/tests/emit_tests.rs b/crates/encore_compiler/tests/emit_tests.rs index 38cdb67..eb85523 100644 --- a/crates/encore_compiler/tests/emit_tests.rs +++ b/crates/encore_compiler/tests/emit_tests.rs @@ -142,3 +142,31 @@ fn test_extern_stub_and_function() { assert_eq!(code[15], opcode::FIN); assert_eq!(code[16], X01); } + +#[test] +fn test_int_literal_at_bounds() { + for n in [(1 << 23) - 1, -(1 << 23)] { + let expr = Expr::Let(X01, Val::Int(n), Box::new(Expr::Fin(X01))); + let mut emitter = Emitter::new(); + emitter.emit_toplevel(&expr); + let code = emitter.into_bytes(); + let bits = n as u32; + assert_eq!(code, [opcode::INT, X01, bits as u8, (bits >> 8) as u8, (bits >> 16) as u8, opcode::FIN, X01]); + } +} + +#[test] +#[should_panic(expected = "out of 24-bit range")] +fn test_int_literal_above_range_rejected() { + let expr = Expr::Let(X01, Val::Int(1 << 23), Box::new(Expr::Fin(X01))); + let mut emitter = Emitter::new(); + emitter.emit_toplevel(&expr); +} + +#[test] +#[should_panic(expected = "out of 24-bit range")] +fn test_int_literal_below_range_rejected() { + let expr = Expr::Let(X01, Val::Int(-(1 << 23) - 1), Box::new(Expr::Fin(X01))); + let mut emitter = Emitter::new(); + emitter.emit_toplevel(&expr); +} diff --git a/crates/encore_compiler/tests/simplify_tests.rs b/crates/encore_compiler/tests/simplify_tests.rs index 2344da0..59eb3bf 100644 --- a/crates/encore_compiler/tests/simplify_tests.rs +++ b/crates/encore_compiler/tests/simplify_tests.rs @@ -207,6 +207,50 @@ fn test_const_fold_mul() { assert_eq!(result, expected); } +#[test] +fn test_const_fold_add_at_max() { + let expr = let_("a", int((1 << 23) - 2), + let_("b", int(1), + let_("c", prim(PrimOp::Int(IntOp::Add), &["a", "b"]), + fin("c")))); + let result = constant_fold(expr); + let expected = let_("a", int((1 << 23) - 2), + let_("b", int(1), + let_("c", int((1 << 23) - 1), + fin("c")))); + assert_eq!(result, expected); +} + +#[test] +fn test_const_fold_add_overflow_not_folded() { + let expr = let_("a", int((1 << 23) - 1), + let_("b", int(1), + let_("c", prim(PrimOp::Int(IntOp::Add), &["a", "b"]), + fin("c")))); + let result = constant_fold(expr.clone()); + assert_eq!(result, expr); +} + +#[test] +fn test_const_fold_sub_underflow_not_folded() { + let expr = let_("a", int(-(1 << 23)), + let_("b", int(1), + let_("c", prim(PrimOp::Int(IntOp::Sub), &["a", "b"]), + fin("c")))); + let result = constant_fold(expr.clone()); + assert_eq!(result, expr); +} + +#[test] +fn test_const_fold_mul_overflow_not_folded() { + let expr = let_("a", int(1 << 12), + let_("b", int(1 << 12), + let_("c", prim(PrimOp::Int(IntOp::Mul), &["a", "b"]), + fin("c")))); + let result = constant_fold(expr.clone()); + assert_eq!(result, expr); +} + #[test] fn test_const_fold_chained() { // let a = 2 in let b = 3 in let c = add(a, b) in let d = mul(c, a) in fin d diff --git a/crates/encore_vm/src/error.rs b/crates/encore_vm/src/error.rs index 985426f..61df5c8 100644 --- a/crates/encore_vm/src/error.rs +++ b/crates/encore_vm/src/error.rs @@ -46,6 +46,7 @@ pub enum VmError { InvalidOpcode { opcode: u8, pc: u16 }, MatchFail { tag: u8, pc: u16 }, ByteRange { value: i32, pc: u16 }, + IntOverflow { pc: u16 }, BadMagic, Truncated, Extern { error: ExternError, slot: u16, pc: u16 }, @@ -59,6 +60,7 @@ impl VmError { VmError::InvalidOpcode { .. } => "invalid opcode", VmError::MatchFail { .. } => "match failure", VmError::ByteRange { .. } => "byte range error", + VmError::IntOverflow { .. } => "integer overflow", VmError::BadMagic => "bad magic", VmError::Truncated => "truncated program", VmError::Extern { .. } => "extern call failed", @@ -77,6 +79,8 @@ impl fmt::Display for VmError { write!(f, "match failure: no branch for tag {tag} at pc=0x{pc:04x}"), VmError::ByteRange { value, pc } => write!(f, "byte range error: value {value} out of 0..255 at pc=0x{pc:04x}"), + VmError::IntOverflow { pc } => + write!(f, "integer overflow: result out of 24-bit range at pc=0x{pc:04x}"), VmError::BadMagic => write!(f, "bad magic bytes (expected ENCR)"), VmError::Truncated => write!(f, "truncated program"), VmError::Extern { error, slot, pc } => diff --git a/crates/encore_vm/src/value.rs b/crates/encore_vm/src/value.rs index 4072acb..de1cc91 100644 --- a/crates/encore_vm/src/value.rs +++ b/crates/encore_vm/src/value.rs @@ -11,6 +11,16 @@ const TYP_BYTES_HDR: u32 = 7; const GC_MARK_BIT: u32 = 0x80 << 8; +/// Smallest integer representable in a `Value` (24-bit signed payload). +pub const INT_MIN: i32 = -(1 << 23); +/// Largest integer representable in a `Value` (24-bit signed payload). +pub const INT_MAX: i32 = (1 << 23) - 1; + +/// Whether `n` fits in the 24-bit signed payload of an integer `Value`. +pub const fn int_in_range(n: i32) -> bool { + n >= INT_MIN && n <= INT_MAX +} + #[derive(Clone, Copy)] pub struct HeapAddress(u16); diff --git a/crates/encore_vm/src/vm.rs b/crates/encore_vm/src/vm.rs index 5332790..a00d397 100644 --- a/crates/encore_vm/src/vm.rs +++ b/crates/encore_vm/src/vm.rs @@ -8,7 +8,7 @@ use crate::program::Program; use crate::registers::Registers; #[cfg(feature = "stats")] use crate::stats::{Clock, VmStats}; -use crate::value::{CodeAddress, GlobalAddress, HeapAddress, Reg, Value}; +use crate::value::{int_in_range, CodeAddress, GlobalAddress, HeapAddress, Reg, Value}; const SELF: Reg = Reg::new(0); const CONT: Reg = Reg::new(1); @@ -455,7 +455,10 @@ impl<'a> Vm<'a> { let rb = self.code.read_reg(); let a = self.registers[ra].int_value()?; let b = self.registers[rb].int_value()?; - self.registers[rd] = Value::int(a.wrapping_add(b)); + match a.checked_add(b) { + Some(n) if int_in_range(n) => self.registers[rd] = Value::int(n), + _ => return Err(VmError::IntOverflow { pc }), + } } opcode::INT_SUB => { @@ -464,7 +467,10 @@ impl<'a> Vm<'a> { let rb = self.code.read_reg(); let a = self.registers[ra].int_value()?; let b = self.registers[rb].int_value()?; - self.registers[rd] = Value::int(a.wrapping_sub(b)); + match a.checked_sub(b) { + Some(n) if int_in_range(n) => self.registers[rd] = Value::int(n), + _ => return Err(VmError::IntOverflow { pc }), + } } opcode::INT_MUL => { @@ -473,7 +479,10 @@ impl<'a> Vm<'a> { let rb = self.code.read_reg(); let a = self.registers[ra].int_value()?; let b = self.registers[rb].int_value()?; - self.registers[rd] = Value::int(a.wrapping_mul(b)); + match a.checked_mul(b) { + Some(n) if int_in_range(n) => self.registers[rd] = Value::int(n), + _ => return Err(VmError::IntOverflow { pc }), + } } opcode::INT_EQ => { diff --git a/crates/encore_vm/tests/vm_tests.rs b/crates/encore_vm/tests/vm_tests.rs index 4eda515..7389cba 100644 --- a/crates/encore_vm/tests/vm_tests.rs +++ b/crates/encore_vm/tests/vm_tests.rs @@ -235,6 +235,32 @@ fn test_int_lt_false() { assert_eq!(result.ctor_tag(), 0); } +#[test] +fn test_int_add_at_max() { + // (2^23 - 2) + 1 = 2^23 - 1 + let code = [ + INT, X01, 0xFE, 0xFF, 0x7F, + INT_1, X02, + INT_ADD, X03, X01, X02, + FIN, X03, + ]; + let result = run(&code, &[]).unwrap(); + assert_eq!(result.int_value().unwrap(), (1 << 23) - 1); +} + +#[test] +fn test_int_sub_at_min() { + // (-2^23 + 1) - 1 = -2^23 + let code = [ + INT, X01, 0x01, 0x00, 0x80, + INT_1, X02, + INT_SUB, X03, X01, X02, + FIN, X03, + ]; + let result = run(&code, &[]).unwrap(); + assert_eq!(result.int_value().unwrap(), -(1 << 23)); +} + // -- Error tests -- #[test] @@ -249,6 +275,70 @@ fn test_heap_overflow() { assert!(matches!(result, Err(VmError::HeapOverflow))); } +#[test] +fn test_int_add_overflow() { + // (2^23 - 1) + 1 + let code = [ + INT, X01, 0xFF, 0xFF, 0x7F, + INT_1, X02, + INT_ADD, X03, X01, X02, + FIN, X03, + ]; + let result = run(&code, &[]); + assert!(matches!(result, Err(VmError::IntOverflow { pc: 7 }))); +} + +#[test] +fn test_int_sub_overflow() { + // -2^23 - 1 + let code = [ + INT, X01, 0x00, 0x00, 0x80, + INT_1, X02, + INT_SUB, X03, X01, X02, + FIN, X03, + ]; + let result = run(&code, &[]); + assert!(matches!(result, Err(VmError::IntOverflow { pc: 7 }))); +} + +#[test] +fn test_int_mul_overflow() { + // 2^12 * 2^11 = 2^23 + let code = [ + INT, X01, 0x00, 0x10, 0x00, + INT, X02, 0x00, 0x08, 0x00, + INT_MUL, X03, X01, X02, + FIN, X03, + ]; + let result = run(&code, &[]); + assert!(matches!(result, Err(VmError::IntOverflow { pc: 10 }))); +} + +#[test] +fn test_int_mul_overflow_i32() { + // (2^23 - 1)^2 overflows i32 too + let code = [ + INT, X01, 0xFF, 0xFF, 0x7F, + INT_MUL, X03, X01, X01, + FIN, X03, + ]; + let result = run(&code, &[]); + assert!(matches!(result, Err(VmError::IntOverflow { pc: 5 }))); +} + +#[test] +fn test_int_mul_at_min() { + // -2^12 * 2^11 = -2^23 + let code = [ + INT, X01, 0x00, 0xF0, 0xFF, + INT, X02, 0x00, 0x08, 0x00, + INT_MUL, X03, X01, X02, + FIN, X03, + ]; + let result = run(&code, &[]).unwrap(); + assert_eq!(result.int_value().unwrap(), -(1 << 23)); +} + #[test] fn test_invalid_opcode() { let code = [0xF0];