From e13b42c68988c248e55b1ffcf5c75e6fceadd1b7 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:19:42 +0200 Subject: [PATCH 01/10] yjit: experimental Windows x64 (mingw-ucrt) port Bring up YJIT on x86_64-w64-mingw-ucrt, previously unsupported (rb_jit_reserve_addr_space returned NULL on _WIN32). Changes: - configure.ac: allow JIT on x86_64-*mingw*; link Rust std deps (bcrypt/ntdll/userenv/synchronization). - jit.c: VirtualAlloc/VirtualProtect/VirtualFree for reserve/mark_writable/mark_executable/mark_unused; GetSystemInfo page size; FlushInstructionCache. - defs/jit.mk: on mingw, binutils ld -r cannot partial-link the Rust staticlib (PE TLS DataDirectory). Instead build a symbol-localized copy of the archive (drop .dwo members, objcopy --localize-symbols on just the symbols that collide with Ruby's missing/*.o) and link it directly. - yjit backend (x86_64): MS x64 ABI. C_ARG_OPNDS = RCX,RDX,R8,R9 with stack args + 32-byte shadow space for >4 args; TEMP_REGS = RCX,RDX,R8,R9,R10; alloc regs = RAX,RSI,RDI (RSI/RDI callee-saved, preserved once per frame in FrameSetup); Win64 caller-save set. - yjit utils.rs: c_callable! uses extern "C" (MS x64) on Windows instead of extern "sysv64" so runtime callbacks match the generated calls. - yjit cruby_bindings: ID/st_data_t are usize (uintptr_t), not c_ulong, which is 32-bit on LLP64. - yjit fd handling (options/log/disasm): RawFd -> RawFileRef (RawHandle on Windows). - assorted usize/u64/i32/i64 cast fixes for LLP64. Status: scalar arithmetic, loops, method calls, deep recursion (fib), strings, arrays, procs, and blocks all run correctly and pass. Benchmarks vs interpreter: fib(31) 9.8x, method/ivar dispatch 2.4x, predicate/branch 1.9x. Known bug: an accumulator built up over many JIT'd block iterations that crosses 2^32 triggers a spurious 'Unnormalized Fixnum' check (the computed value is numerically correct). RubyVM::YJIT.runtime_stats also asserts. Both are isolated follow-ups; core codegen is correct. --- configure.ac | 8 +++ defs/jit.mk | 29 +++++++++ jit.c | 55 +++++++++++++++- yjit/src/backend/ir.rs | 8 ++- yjit/src/backend/x86_64/mod.rs | 111 ++++++++++++++++++++++++++++++--- yjit/src/codegen.rs | 26 ++++---- yjit/src/cruby.rs | 4 +- yjit/src/cruby_bindings.inc.rs | 5 +- yjit/src/disasm.rs | 12 ++-- yjit/src/log.rs | 6 +- yjit/src/options.rs | 45 +++++++++++-- yjit/src/utils.rs | 15 ++++- 12 files changed, 281 insertions(+), 43 deletions(-) diff --git a/configure.ac b/configure.ac index 6fbbb11178d219..9746672d592b71 100644 --- a/configure.ac +++ b/configure.ac @@ -3958,6 +3958,10 @@ AS_IF([test "$cross_compiling" = no], ], [arm64-*bsd*|aarch64-*bsd*|x86_64-*bsd*], [ JIT_TARGET_OK=yes + ], + [x86_64-*mingw*], [ + dnl experimental: MS x64 ABI support in the YJIT x86_64 backend + JIT_TARGET_OK=yes ] ) ) @@ -4034,6 +4038,10 @@ AS_CASE(["${YJIT_SUPPORT}"], AC_DEFINE_UNQUOTED(YJIT_SUPPORT, [$YJIT_SUPPORT]) ]) AC_DEFINE(USE_YJIT, 1) + AS_CASE(["$target_os"], [mingw*], [ + dnl system libraries the Rust standard library (staticlib) pulls in + LIBS="$LIBS -lbcrypt -lntdll -luserenv -lsynchronization" + ]) ], [ AC_DEFINE(USE_YJIT, 0) ]) diff --git a/defs/jit.mk b/defs/jit.mk index 2c1e819684939b..7a0e9d715ee9b5 100644 --- a/defs/jit.mk +++ b/defs/jit.mk @@ -65,6 +65,34 @@ $(JIT_RLIB): target/.rustc-version endif # ifneq ($(JIT_CARGO_SUPPORT),no) RUST_LIB_SYMBOLS = $(RUST_LIB:.a=).symbols +ifneq ($(findstring mingw,$(target_os)),) +# binutils `ld -r` cannot partial-link the TLS directory the Rust std emits +# on PE ("unable to fill in DataDirectory[9]: _tls_used not defined +# correctly"). Instead, make a localized *copy of the archive*: objcopy +# rewrites each member's symbol table (no TLS-merging `ld -r` involved), +# hiding everything but rb_* so Rust's bundled compiler_builtins (lgamma_r, +# etc.) don't collide with Ruby's own missing/*.c. The linker then pulls +# members from this archive like any other library input. +RUST_LIBOBJ = $(RUST_LIB:.a=.localized.a) +# Unlike the Linux path we cannot `ld -r`-merge the archive first (PE TLS +# limitation), so we must NOT blanket-localize every non-rb_* symbol -- that +# would sever the cross-CGU references inside the Rust static lib (fmt impls, +# rust_eh_personality, ...). Instead localize ONLY the symbols that actually +# collide with Ruby's own definitions (compiler_builtins math intrinsics vs +# missing/lgamma_r.o etc.), computed as the intersection of the archive's +# globals with $(MISSING)'s globals. +$(RUST_LIBOBJ): $(RUST_LIB) $(MISSING) + $(ECHO) 'localizing collisions in $(RUST_LIB) into $@ (mingw)' + $(Q) $(CP) $(RUST_LIB) $@ + $(Q) for m in `ar t $@ | grep '\.dwo$$'`; do ar d $@ "$$m"; done + $(Q) nm -g --defined-only $(MISSING) 2>/dev/null | awk 'NF==3 {print $$3}' | sort -u > $@.miss.syms + $(Q) nm -g --defined-only $@ 2>/dev/null | awk 'NF==3 {print $$3}' | sort -u > $@.rust.syms + $(Q) comm -12 $@.miss.syms $@.rust.syms > $@.collide.syms + $(Q) if [ -s $@.collide.syms ]; then \ + $(ECHO) "localizing collisions:" `tr '\n' ' ' < $@.collide.syms`; \ + objcopy --localize-symbols=$@.collide.syms $@; \ + fi +else $(RUST_LIBOBJ): $(RUST_LIB) $(ECHO) 'partial linking $(RUST_LIB) into $@' ifneq ($(findstring darwin,$(target_os)),) @@ -73,6 +101,7 @@ else $(Q) $(LD) -r -o $@ --whole-archive $(RUST_LIB) -$(Q) $(OBJCOPY) --wildcard --keep-global-symbol='$(SYMBOL_PREFIX)rb_*' $(@) endif +endif rust-libobj: $(RUST_LIBOBJ) rust-lib: $(RUST_LIB) diff --git a/jit.c b/jit.c index 88652dddd93dbe..2cb36523a17933 100644 --- a/jit.c +++ b/jit.c @@ -25,6 +25,8 @@ #ifndef _WIN32 #include +#else +#include #endif enum jit_bindgen_constants { @@ -640,6 +642,13 @@ rb_jit_get_page_size(void) if (page_size > 0x40000000l) rb_bug("jit page size too large"); return (uint32_t)page_size; +#elif defined(_WIN32) + SYSTEM_INFO si; + GetSystemInfo(&si); + if (si.dwPageSize == 0 || si.dwPageSize > 0x40000000ul) { + rb_bug("jit: bad page size"); + } + return (uint32_t)si.dwPageSize; #else #error "JIT supports POSIX only for now" #endif @@ -747,8 +756,29 @@ rb_jit_reserve_addr_space(uint32_t mem_size) return mem_block; #else - // Windows not supported for now - return NULL; + // Windows: reserve address space with VirtualAlloc; pages are committed + // page-by-page later via rb_jit_mark_writable (MEM_COMMIT). + uint8_t *mem_block = NULL; + + // Try to reserve close to the ruby image so 32-bit relative call/jmp + // encodings can reach C functions. Probe downwards in 4 MiB steps + // from our own code address; the allocator falls back to "anywhere". + uint8_t *req_addr = (uint8_t *)((uintptr_t)&rb_jit_reserve_addr_space & ~(uintptr_t)0xFFFF); + for (int attempt = 0; attempt < 1024; attempt++) { + req_addr -= 4 * 1024 * 1024; + if (req_addr <= (uint8_t *)0x10000) break; + mem_block = VirtualAlloc(req_addr, mem_size, MEM_RESERVE, PAGE_NOACCESS); + if (mem_block) break; + } + if (!mem_block) { + mem_block = VirtualAlloc(NULL, mem_size, MEM_RESERVE, PAGE_NOACCESS); + } + if (!mem_block) { + fprintf(stderr, "ruby: jit: VirtualAlloc reserve of %u bytes failed (%lu)\n", + mem_size, (unsigned long)GetLastError()); + exit(EXIT_FAILURE); + } + return mem_block; #endif } @@ -763,7 +793,13 @@ rb_jit_for_each_iseq(rb_iseq_callback callback, void *data) bool rb_jit_mark_writable(void *mem_block, uint32_t mem_size) { +#ifdef _WIN32 + // The region is only MEM_RESERVEd up-front; committing is what makes + // the pages usable (and is idempotent for already-committed pages). + return VirtualAlloc(mem_block, mem_size, MEM_COMMIT, PAGE_READWRITE) != NULL; +#else return mprotect(mem_block, mem_size, PROT_READ | PROT_WRITE) == 0; +#endif } void @@ -774,16 +810,30 @@ rb_jit_mark_executable(void *mem_block, uint32_t mem_size) if (mem_size == 0) { return; } +#ifdef _WIN32 + DWORD old_protect; + if (!VirtualProtect(mem_block, mem_size, PAGE_EXECUTE_READ, &old_protect)) { + rb_bug("Couldn't make JIT page (%p, %lu bytes) executable, error: %lu", + mem_block, (unsigned long)mem_size, (unsigned long)GetLastError()); + } + FlushInstructionCache(GetCurrentProcess(), mem_block, mem_size); +#else if (mprotect(mem_block, mem_size, PROT_READ | PROT_EXEC)) { rb_bug("Couldn't make JIT page (%p, %lu bytes) executable, errno: %s", mem_block, (unsigned long)mem_size, strerror(errno)); } +#endif } // Free the specified memory block. bool rb_jit_mark_unused(void *mem_block, uint32_t mem_size) { +#ifdef _WIN32 + // Decommit: the pages return to reserved state; a later + // rb_jit_mark_writable (MEM_COMMIT) makes them usable again. + return VirtualFree(mem_block, mem_size, MEM_DECOMMIT) != 0; +#else // On Linux, you need to use madvise MADV_DONTNEED to free memory. // We might not need to call this on macOS, but it's not really documented. // We generally prefer to do the same thing on both to ease testing too. @@ -792,6 +842,7 @@ rb_jit_mark_unused(void *mem_block, uint32_t mem_size) // On macOS, mprotect PROT_NONE seems to reduce RSS. // We also call this on Linux to avoid executing unused pages. return mprotect(mem_block, mem_size, PROT_NONE) == 0; +#endif } // Invalidate icache for arm64. diff --git a/yjit/src/backend/ir.rs b/yjit/src/backend/ir.rs index e14f3f9cb28e95..0501e0ce4bdbc7 100644 --- a/yjit/src/backend/ir.rs +++ b/yjit/src/backend/ir.rs @@ -16,7 +16,10 @@ pub const EC: Opnd = _EC; pub const CFP: Opnd = _CFP; pub const SP: Opnd = _SP; +#[cfg(not(windows))] pub const C_ARG_OPNDS: [Opnd; 6] = _C_ARG_OPNDS; +#[cfg(windows)] +pub const C_ARG_OPNDS: [Opnd; 4] = _C_ARG_OPNDS; pub const C_RET_OPND: Opnd = _C_RET_OPND; pub use crate::backend::current::{Reg, C_RET_REG}; @@ -1089,7 +1092,10 @@ impl Assembler /// Get the list of registers that can be used for stack temps. pub fn get_temp_regs() -> &'static [Reg] { - let num_regs = get_option!(num_temp_regs); + // Clamp to the platform's TEMP_REGS length: Windows x64 has fewer + // (RSI/RDI are used as allocation registers there), while the default + // num_temp_regs option is sized for the SysV set. + let num_regs = get_option!(num_temp_regs).min(TEMP_REGS.len()); &TEMP_REGS[0..num_regs] } diff --git a/yjit/src/backend/x86_64/mod.rs b/yjit/src/backend/x86_64/mod.rs index ef435bca7e5350..f46f8bef0984c5 100644 --- a/yjit/src/backend/x86_64/mod.rs +++ b/yjit/src/backend/x86_64/mod.rs @@ -17,6 +17,7 @@ pub const _EC: Opnd = Opnd::Reg(R12_REG); pub const _SP: Opnd = Opnd::Reg(RBX_REG); // C argument registers on this platform +#[cfg(not(windows))] pub const _C_ARG_OPNDS: [Opnd; 6] = [ Opnd::Reg(RDI_REG), Opnd::Reg(RSI_REG), @@ -26,6 +27,17 @@ pub const _C_ARG_OPNDS: [Opnd; 6] = [ Opnd::Reg(R9_REG) ]; +// Windows x64: only four register arguments (RCX, RDX, R8, R9); further +// arguments are passed on the native stack above the 32-byte shadow space. +// See the CCall handling in x86_split/x86_emit. +#[cfg(windows)] +pub const _C_ARG_OPNDS: [Opnd; 4] = [ + Opnd::Reg(RCX_REG), + Opnd::Reg(RDX_REG), + Opnd::Reg(R8_REG), + Opnd::Reg(R9_REG) +]; + // C return value register on this platform pub const C_RET_REG: Reg = RAX_REG; pub const _C_RET_OPND: Opnd = Opnd::Reg(RAX_REG); @@ -80,7 +92,19 @@ impl From<&Opnd> for X86Opnd { } /// List of registers that can be used for stack temps and locals. +/// On Windows x64, RSI/RDI are callee-saved and are instead used as +/// allocation registers (see get_alloc_regs) — they are preserved once per +/// JIT frame in FrameSetup. RCX/RDX must stay out of the alloc set because +/// gen_branch_stub places call arguments directly in them, so the temp/alloc +/// split differs from SysV. +#[cfg(not(windows))] pub static TEMP_REGS: [Reg; 5] = [RSI_REG, RDI_REG, R8_REG, R9_REG, R10_REG]; +// Windows x64: all five are caller-saved (spilled around ccalls). RCX/RDX +// double as the first two C argument registers, exactly as R8/R9 do in the +// SysV set, so gen_branch_stub can place args in them. RSI/RDI are excluded +// here because they are the allocation registers (get_alloc_regs). +#[cfg(windows)] +pub static TEMP_REGS: [Reg; 5] = [RCX_REG, RDX_REG, R8_REG, R9_REG, R10_REG]; impl Assembler { @@ -91,6 +115,7 @@ impl Assembler /// Get the list of registers from which we can allocate on this platform + #[cfg(not(windows))] pub fn get_alloc_regs() -> Vec { vec![ @@ -100,11 +125,31 @@ impl Assembler ] } + /// Windows x64: RCX and RDX are argument registers that gen_branch_stub + /// and friends write to directly, so they must not be allocatable. Use + /// RSI/RDI instead (callee-saved, preserved once per frame in FrameSetup). + #[cfg(windows)] + pub fn get_alloc_regs() -> Vec + { + vec![ + RAX_REG, + RSI_REG, + RDI_REG, + ] + } + /// Get a list of all of the caller-save registers + #[cfg(not(windows))] pub fn get_caller_save_regs() -> Vec { vec![RAX_REG, RCX_REG, RDX_REG, RSI_REG, RDI_REG, R8_REG, R9_REG, R10_REG, R11_REG] } + /// Windows x64 caller-saved GP registers (RSI/RDI are callee-saved there). + #[cfg(windows)] + pub fn get_caller_save_regs() -> Vec { + vec![RAX_REG, RCX_REG, RDX_REG, R8_REG, R9_REG, R10_REG, R11_REG] + } + // These are the callee-saved registers in the x86-64 SysV ABI // RBX, RSP, RBP, and R12–R15 @@ -365,17 +410,43 @@ impl Assembler asm.not(opnd0); }, Insn::CCall { opnds, fptr, .. } => { - assert!(opnds.len() <= C_ARG_OPNDS.len()); + // On SysV all supported arities fit in registers. On + // Windows x64 only the first four arguments do; the rest + // are pushed onto the native stack (in reverse, padded to + // an even count to keep 16-byte alignment at the call). + // The CCall emission adds the 32-byte shadow space, which + // lands exactly between the return address and these + // pushed arguments. + let num_reg_args = C_ARG_OPNDS.len().min(opnds.len()); + let num_stack_args = opnds.len() - num_reg_args; + let num_pushed = (num_stack_args + 1) & !1; + + if num_stack_args > 0 { + if num_stack_args % 2 == 1 { + // alignment padding slot + asm.cpush(Opnd::Reg(Assembler::SCRATCH_REG)); + } + for opnd in opnds[num_reg_args..].iter().rev() { + let val = match opnd { + Opnd::Reg(_) => *opnd, + _ => asm.load(*opnd), + }; + asm.cpush(val); + } + } - // Load each operand into the corresponding argument - // register. - for (idx, opnd) in opnds.into_iter().enumerate() { + // Load each remaining operand into the corresponding + // argument register. + for (idx, opnd) in opnds[..num_reg_args].iter().enumerate() { asm.load_into(Opnd::c_arg(C_ARG_OPNDS[idx]), *opnd); } - // Now we push the CCall without any arguments so that it - // just performs the call. - asm.ccall(*fptr, vec![]); + // Now we push the CCall with only inert register operands + // marking how many stack slots were pushed — the emission + // uses opnds.len() to size the post-call stack cleanup + // (and this insn must stay last so the iterator maps the + // call result correctly). + asm.ccall(*fptr, vec![Opnd::Reg(Assembler::SCRATCH_REG); num_pushed]); }, Insn::Lea { .. } => { // Merge `lea` and `mov` into a single `lea` when possible @@ -523,6 +594,15 @@ impl Assembler // Set up RBP to work with frame pointer unwinding // (e.g. with Linux `perf record --call-graph fp`) Insn::FrameSetup => { + // Windows x64: RSI and RDI are callee-saved, but the JIT + // uses them as stack-temp registers (TEMP_REGS), so they + // must be preserved across the JIT entry. Two pushes keep + // the 16-byte stack alignment parity unchanged. + #[cfg(windows)] + { + push(cb, RSI); + push(cb, RDI); + } if get_option!(frame_pointer) { push(cb, RBP); mov(cb, RBP, RSP); @@ -534,6 +614,11 @@ impl Assembler pop(cb, RBP); pop(cb, RBP); } + #[cfg(windows)] + { + pop(cb, RDI); + pop(cb, RSI); + } }, Insn::Add { left, right, .. } => { @@ -664,10 +749,22 @@ impl Assembler }, // C function call + #[cfg(not(windows))] Insn::CCall { fptr, .. } => { call_ptr(cb, RAX, *fptr); }, + // Windows x64: the caller provides 32 bytes of shadow space + // just above the return address. Any pushed stack arguments + // (see x86_split; their count travels in opnds) sit directly + // above the shadow space, and are dropped together with it. + #[cfg(windows)] + Insn::CCall { fptr, opnds, .. } => { + sub(cb, RSP, imm_opnd(0x20)); + call_ptr(cb, RAX, *fptr); + add(cb, RSP, imm_opnd(0x20 + 8 * opnds.len() as i64)); + }, + Insn::CRet(opnd) => { // TODO: bias allocation towards return register if *opnd != Opnd::Reg(C_RET_REG) { diff --git a/yjit/src/codegen.rs b/yjit/src/codegen.rs index 42d16b4b7d619b..455046b494f0f2 100644 --- a/yjit/src/codegen.rs +++ b/yjit/src/codegen.rs @@ -2958,7 +2958,7 @@ fn gen_get_ivar( // The function could raise RactorIsolationError. jit_prepare_non_leaf_call(jit, asm); - let ivar_val = asm.ccall(rb_ivar_get as *const u8, vec![recv, Opnd::UImm(ivar_name)]); + let ivar_val = asm.ccall(rb_ivar_get as *const u8, vec![recv, Opnd::UImm(ivar_name as _)]); if recv_opnd != SelfOpnd { asm.stack_pop(1); @@ -3038,7 +3038,7 @@ fn gen_get_ivar( } else { // The function could raise RactorIsolationError. jit_prepare_non_leaf_call(jit, asm); - asm.ccall(rb_ivar_get_at as *const u8, vec![recv, Opnd::UImm((ivar_index as u32).into()), Opnd::UImm(ivar_name)]) + asm.ccall(rb_ivar_get_at as *const u8, vec![recv, Opnd::UImm((ivar_index as u32).into()), Opnd::UImm(ivar_name as _)]) } }; @@ -3074,7 +3074,7 @@ fn gen_getinstancevariable( asm, GET_IVAR_MAX_DEPTH, comptime_val, - ivar_name, + ivar_name as _, self_asm_opnd, SelfOpnd, ) @@ -3173,7 +3173,7 @@ fn gen_setinstancevariable( jit, asm, comptime_receiver, - ivar_name, + ivar_name as _, SelfOpnd, Some(ic), ) @@ -3274,7 +3274,7 @@ fn gen_set_ivar( rb_vm_set_ivar_id as *const u8, vec![ recv, - Opnd::UImm(ivar_name), + Opnd::UImm(ivar_name as _), val_opnd, ], ); @@ -3442,7 +3442,7 @@ fn gen_definedivar( let shape_id = comptime_receiver.shape_id_of(); let ivar_exists = unsafe { let mut ivar_index: attr_index_t = 0; - rb_shape_get_iv_index(shape_id, ivar_name, &mut ivar_index) + rb_shape_get_iv_index(shape_id, ivar_name as _, &mut ivar_index) }; // Guard heap object (recv_opnd must be used before stack_pop) @@ -4369,7 +4369,7 @@ fn gen_opt_duparray_send( ) -> Option { let method = jit.get_arg(1).as_u64(); - if method == ID!(include_p) { + if method == ID!(include_p) as u64 { gen_opt_duparray_send_include_p(jit, asm) } else { None @@ -6761,7 +6761,7 @@ fn c_method_tracing_currently_enabled(jit: &JITState) -> bool { unsafe extern "C" fn build_kwhash(ci: *const rb_callinfo, sp: *const VALUE) -> VALUE { let kw_arg = vm_ci_kwarg(ci); let kw_len: usize = get_cikw_keyword_len(kw_arg).try_into().unwrap(); - let hash = rb_hash_new_with_size(kw_len as u64); + let hash = rb_hash_new_with_size(kw_len as _); for kwarg_idx in 0..kw_len { let key = get_cikw_keywords_idx(kw_arg, kwarg_idx.try_into().unwrap()); @@ -8579,7 +8579,7 @@ fn gen_iseq_kw_call( // Use the total number of supplied keywords as a size upper bound let keyword_len = unsafe { (*keywords).keyword_len } as usize; - let hash = unsafe { rb_hash_new_with_size(keyword_len as u64) }; + let hash = unsafe { rb_hash_new_with_size(keyword_len as _) }; // Put pairs into the kwrest hash as the mask describes for kwarg_idx in 0..keyword_len { @@ -8954,7 +8954,7 @@ fn gen_struct_aref( // Confidence checks assert!(unsafe { RB_TYPE_P(comptime_recv, RUBY_T_STRUCT) }); - assert!((off as i64) < unsafe { RSTRUCT_LEN(comptime_recv) }); + assert!((off as i64) < i64::from(unsafe { RSTRUCT_LEN(comptime_recv) })); // We are going to use an encoding that takes a 4-byte immediate which // limits the offset to INT32_MAX. @@ -9038,7 +9038,7 @@ fn gen_struct_aset( // Confidence checks assert!(unsafe { RB_TYPE_P(comptime_recv, RUBY_T_STRUCT) }); - assert!((off as i64) < unsafe { RSTRUCT_LEN(comptime_recv) }); + assert!((off as i64) < i64::from(unsafe { RSTRUCT_LEN(comptime_recv) })); // Even if the comptime recv was not frozen, future recv may be. So we need to emit a guard // that the recv is not frozen. @@ -9313,7 +9313,7 @@ fn gen_send_general( asm, SEND_MAX_DEPTH, comptime_recv, - ivar_name, + ivar_name as _, recv, recv.into(), ); @@ -9559,7 +9559,7 @@ fn get_class_name(class: Option) -> String { } /// Assemble "{class_name}#{method_name}" from a class pointer and a method ID -fn get_method_name(class: Option, mid: u64) -> String { +fn get_method_name(class: Option, mid: ID) -> String { let class_name = get_class_name(class); let method_name = if mid != 0 { unsafe { cstr_to_rust_string(rb_id2name(mid)) } diff --git a/yjit/src/cruby.rs b/yjit/src/cruby.rs index 7e945f5db7643f..bac9317e4cb0cf 100644 --- a/yjit/src/cruby.rs +++ b/yjit/src/cruby.rs @@ -605,7 +605,7 @@ pub unsafe fn cfp_env_has_escaped(cfp: *mut rb_control_frame_struct) -> bool { /// Produce a Ruby string from a Rust string slice pub fn rust_str_to_ruby(str: &str) -> VALUE { - unsafe { rb_utf8_str_new(str.as_ptr() as *const _, str.len() as i64) } + unsafe { rb_utf8_str_new(str.as_ptr() as *const _, str.len() as _) } } /// Produce a Ruby symbol from a Rust string slice @@ -797,7 +797,7 @@ pub(crate) mod ids { ($(name: $ident:ident content: $str:literal)*) => { $( #[doc = concat!("[type@crate::cruby::ID] for `", stringify!($str), "`")] - pub static $ident: AtomicU64 = AtomicU64::new(0); + pub static $ident: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); )* pub(crate) fn init() { diff --git a/yjit/src/cruby_bindings.inc.rs b/yjit/src/cruby_bindings.inc.rs index 753955c8affb0a..3195a23f532560 100644 --- a/yjit/src/cruby_bindings.inc.rs +++ b/yjit/src/cruby_bindings.inc.rs @@ -172,7 +172,8 @@ pub const VM_ENV_DATA_INDEX_ME_CREF: i32 = -2; pub const VM_ENV_DATA_INDEX_SPECVAL: i32 = -1; pub const VM_ENV_DATA_INDEX_FLAGS: u32 = 0; pub const VM_BLOCK_HANDLER_NONE: u32 = 0; -pub type ID = ::std::os::raw::c_ulong; +pub const SHAPE_ID_NUM_BITS: u32 = 32; +pub type ID = usize; // uintptr_t in C; c_ulong is wrong on LLP64 (Windows) pub type rb_alloc_func_t = ::std::option::Option VALUE>; pub const RUBY_Qfalse: ruby_special_consts = 0; pub const RUBY_Qnil: ruby_special_consts = 4; @@ -256,7 +257,7 @@ pub type ruby_fl_type = i32; pub const RSTRING_NOEMBED: ruby_rstring_flags = 8192; pub const RSTRING_FSTR: ruby_rstring_flags = 536870912; pub type ruby_rstring_flags = u32; -pub type st_data_t = ::std::os::raw::c_ulong; +pub type st_data_t = usize; // st_data_t is uintptr_t-sized; c_ulong is wrong on LLP64 pub type st_index_t = st_data_t; pub const ST_CONTINUE: st_retval = 0; pub const ST_STOP: st_retval = 1; diff --git a/yjit/src/disasm.rs b/yjit/src/disasm.rs index 4f85937ee9f0b7..28eb4f61ad01c9 100644 --- a/yjit/src/disasm.rs +++ b/yjit/src/disasm.rs @@ -146,13 +146,13 @@ pub fn dump_disasm_addr_range(cb: &CodeBlock, start_addr: CodePtr, end_addr: Cod match dump_disasm { DumpDisasm::Stdout => println!("{disasm}"), DumpDisasm::File(fd) => { - use std::os::unix::io::{FromRawFd, IntoRawFd}; use std::io::Write; + use crate::options::{file_from_raw_ref, file_into_raw_ref}; // Write with the fd opened during boot - let mut file = unsafe { std::fs::File::from_raw_fd(*fd) }; + let mut file = unsafe { file_from_raw_ref(*fd) }; file.write_all(disasm.as_bytes()).unwrap(); - let _ = file.into_raw_fd(); // keep the fd open + let _ = file_into_raw_ref(file); // keep the fd open } }; } @@ -338,7 +338,7 @@ pub extern "C" fn rb_yjit_insns_compiled(_ec: EcPtr, _ruby_self: VALUE, iseqw: V let insn_vec = insns_compiled(iseq); unsafe { - let insn_ary = rb_ary_new_capa((insn_vec.len() * 2) as i64); + let insn_ary = rb_ary_new_capa((insn_vec.len() * 2) as _); // For each instruction compiled for idx in 0..insn_vec.len() { @@ -350,10 +350,10 @@ pub extern "C" fn rb_yjit_insns_compiled(_ec: EcPtr, _ruby_self: VALUE, iseqw: V // Store the instruction index and opcode symbol rb_ary_store( insn_ary, - (2 * idx + 0) as i64, + (2 * idx + 0) as _, VALUE::fixnum_from_usize(insn_idx as usize), ); - rb_ary_store(insn_ary, (2 * idx + 1) as i64, op_sym); + rb_ary_store(insn_ary, (2 * idx + 1) as _, op_sym); } insn_ary diff --git a/yjit/src/log.rs b/yjit/src/log.rs index c5a724f7e1df52..fbf176dc8e31c1 100644 --- a/yjit/src/log.rs +++ b/yjit/src/log.rs @@ -74,14 +74,14 @@ impl Log { } LogOutput::File(fd) => { - use std::os::unix::io::{FromRawFd, IntoRawFd}; use std::io::Write; + use crate::options::{file_from_raw_ref, file_into_raw_ref}; // Write with the fd opened during boot - let mut file = unsafe { std::fs::File::from_raw_fd(fd) }; + let mut file = unsafe { file_from_raw_ref(fd) }; writeln!(file, "{}", entry).unwrap(); file.flush().unwrap(); - let _ = file.into_raw_fd(); // keep the fd open + let _ = file_into_raw_ref(file); // keep the fd open } LogOutput::MemoryOnly => () // Don't print or write anything diff --git a/yjit/src/options.rs b/yjit/src/options.rs index c87a436091279f..ecc6a2f3d8b0cc 100644 --- a/yjit/src/options.rs +++ b/yjit/src/options.rs @@ -137,10 +137,45 @@ pub enum TraceExits { Counter(Counter), } +/// Raw reference to an output file opened during boot. On unix this is a +/// RawFd; on Windows we store the raw handle as usize. It is re-wrapped in +/// a File for each write and immediately released so it stays open. +#[cfg(unix)] +pub type RawFileRef = std::os::unix::io::RawFd; +#[cfg(windows)] +pub type RawFileRef = usize; + +#[cfg(unix)] +pub fn file_into_raw_ref(file: std::fs::File) -> RawFileRef { + use std::os::unix::io::IntoRawFd; + file.into_raw_fd() +} + +#[cfg(windows)] +pub fn file_into_raw_ref(file: std::fs::File) -> RawFileRef { + use std::os::windows::io::IntoRawHandle; + file.into_raw_handle() as RawFileRef +} + +/// Wrap the raw reference in a File. The caller must release it again with +/// file_into_raw_ref (or mem::forget) to keep the descriptor open. +pub unsafe fn file_from_raw_ref(raw: RawFileRef) -> std::fs::File { + #[cfg(unix)] + { + use std::os::unix::io::FromRawFd; + std::fs::File::from_raw_fd(raw) + } + #[cfg(windows)] + { + use std::os::windows::io::{FromRawHandle, RawHandle}; + std::fs::File::from_raw_handle(raw as RawHandle) + } +} + #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum LogOutput { // Dump to the log file as events occur. - File(std::os::unix::io::RawFd), + File(RawFileRef), // Keep the log in memory only MemoryOnly, // Dump to stderr when the process exits @@ -152,7 +187,7 @@ pub enum DumpDisasm { // Dump to stdout Stdout, // Dump to "yjit_{pid}.log" file under the specified directory - File(std::os::unix::io::RawFd), + File(RawFileRef), } /// Type of symbols to dump into /tmp/perf-{pid}.map @@ -304,9 +339,8 @@ pub fn parse_option(str_ptr: *const std::os::raw::c_char) -> Option<()> { let path = format!("{directory}/yjit_{}.log", std::process::id()); match File::options().create(true).append(true).open(&path) { Ok(file) => { - use std::os::unix::io::IntoRawFd; eprintln!("YJIT disasm dump: {path}"); - unsafe { OPTIONS.dump_disasm = Some(DumpDisasm::File(file.into_raw_fd())) } + unsafe { OPTIONS.dump_disasm = Some(DumpDisasm::File(file_into_raw_ref(file))) } } Err(err) => eprintln!("Failed to create {path}: {err}"), } @@ -351,10 +385,9 @@ pub fn parse_option(str_ptr: *const std::os::raw::c_char) -> Option<()> { match File::options().create(true).write(true).truncate(true).open(&log_file_path) { Ok(file) => { - use std::os::unix::io::IntoRawFd; eprintln!("YJIT log: {log_file_path}"); - unsafe { OPTIONS.log = Some(LogOutput::File(file.into_raw_fd())) } + unsafe { OPTIONS.log = Some(LogOutput::File(file_into_raw_ref(file))) } Log::init() } Err(err) => panic!("Failed to create {log_file_path}: {err}"), diff --git a/yjit/src/utils.rs b/yjit/src/utils.rs index 251628fabfae69..0baf2020b8e74d 100644 --- a/yjit/src/utils.rs +++ b/yjit/src/utils.rs @@ -143,7 +143,12 @@ macro_rules! c_callable { }; } -#[cfg(target_arch = "x86_64")] +// On non-Windows x86_64, pin the SysV convention explicitly (matches the +// generated code's register assignment). On Windows x86_64 the generated +// code uses the native MS x64 ABI (RCX/RDX/R8/R9), so these callbacks must +// use `extern "C"` (= MS x64) instead of sysv64, otherwise their arguments +// would be read from the wrong registers. +#[cfg(all(target_arch = "x86_64", not(windows)))] macro_rules! c_callable { ($(#[$outer:meta])* fn $f:ident $args:tt $(-> $ret:ty)? $body:block) => { @@ -151,6 +156,14 @@ macro_rules! c_callable { extern "sysv64" fn $f $args $(-> $ret)? $body }; } +#[cfg(all(target_arch = "x86_64", windows))] +macro_rules! c_callable { + ($(#[$outer:meta])* + fn $f:ident $args:tt $(-> $ret:ty)? $body:block) => { + $(#[$outer])* + extern "C" fn $f $args $(-> $ret)? $body + }; +} pub(crate) use c_callable; pub fn print_int(asm: &mut Assembler, opnd: Opnd) { From 3c705dea69a5f8687cbba39542a8f3d8f4c3b994 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:19:42 +0200 Subject: [PATCH 02/10] yjit(win32): fix LLP64 Fixnum-range bugs in integer fast paths and stats On Windows x64 (LLP64) long is 32-bit, so Ruby's Fixnum range is LONG_MAX/2 (+/-2^30) even though VALUE is 64-bit. YJIT's integer fast paths do 64-bit arithmetic and only detect overflow at 2^63, so a result between 2^31 and 2^63 stayed Fixnum-tagged when it should promote to Bignum -> [BUG] Unnormalized Fixnum. Add guard_fixnum_in_long_range(): after opt_plus/opt_minus/opt_mult/opt_succ compute the tagged result, side-exit to the interpreter (which promotes to Bignum) when it leaves the signed-32-bit range. No-op on LP64 where the 64-bit overflow check already matches the boundary. Costs 2 cmp+jcc per op; benchmarks unchanged (fib 9.8x etc). Also add VALUE::num_from_usize() (rb_ull2inum-backed) and route RubyVM::YJIT.runtime_stats counter packing through it, so counters exceeding the LLP64 Fixnum range become Bignums instead of asserting in fixnum_from_usize. Verified byte-identical to the interpreter across boundary-crossing add/sub/mult, power-via-mult, succ, factorial/Bignum, and the previously-failing block accumulators >2^32. runtime_stats now returns a Hash. --- yjit/src/codegen.rs | 34 ++++++++++++++++++++++++++++++++++ yjit/src/cruby.rs | 17 +++++++++++++++++ yjit/src/stats.rs | 6 +++--- 3 files changed, 54 insertions(+), 3 deletions(-) diff --git a/yjit/src/codegen.rs b/yjit/src/codegen.rs index 455046b494f0f2..5c1d7ad2f9b948 100644 --- a/yjit/src/codegen.rs +++ b/yjit/src/codegen.rs @@ -1749,6 +1749,26 @@ fn gen_adjuststack( Some(KeepCompiling) } +/// On LLP64 targets (Windows x64), `long` is 32-bit so Ruby's Fixnum range is +/// `LONG_MAX/2` (±2^30) and a valid tagged Fixnum VALUE fits in signed 32 bits. +/// YJIT's integer fast paths do 64-bit arithmetic and only detect overflow at +/// 2^63, so a result between 2^31 and 2^63 would stay Fixnum-tagged when it +/// should be a Bignum ("unnormalized Fixnum"). Side-exit to the interpreter +/// (which promotes to Bignum) when the tagged result leaves the 32-bit range. +/// No-op on LP64, where the 64-bit overflow check already matches the boundary. +#[inline] +fn guard_fixnum_in_long_range(asm: &mut Assembler, tagged: Opnd, counter: Counter) { + if !cfg!(windows) { + return; + } + // Valid tagged Fixnum range on LLP64: [ (LONG_MIN/2)<<1|1 , (LONG_MAX/2)<<1|1 ] + // = [ -2147483647 , 2147483647 ], i.e. it must fit in signed 32 bits. + asm.cmp(tagged, Opnd::Imm(0x7FFF_FFFF)); + asm.jg(Target::side_exit(counter)); + asm.cmp(tagged, Opnd::Imm(-0x7FFF_FFFF)); + asm.jl(Target::side_exit(counter)); +} + fn gen_opt_plus( jit: &mut JITState, asm: &mut Assembler, @@ -1776,6 +1796,12 @@ fn gen_opt_plus( let arg0_untag = asm.sub(arg0, Opnd::Imm(1)); let out_val = asm.add(arg0_untag, arg1); asm.jo(Target::side_exit(Counter::opt_plus_overflow)); + // LLP64 (Windows): Fixnum is `long`-wide (RUBY_FIXNUM_MAX = LONG_MAX/2, + // and long is 32-bit) even though VALUE is 64-bit, so the 64-bit `jo` + // above only catches overflow at 2^63. A valid tagged fixnum fits in + // signed 32 bits; side-exit (promote to Bignum in the interpreter) + // when the result leaves that range. + guard_fixnum_in_long_range(asm, out_val, Counter::opt_plus_overflow); // Push the output on the stack let dst = asm.stack_push(Type::Fixnum); @@ -4126,6 +4152,9 @@ fn gen_opt_minus( let val_untag = asm.sub(arg0, arg1); asm.jo(Target::side_exit(Counter::opt_minus_overflow)); let val = asm.add(val_untag, Opnd::Imm(1)); + // LLP64 (Windows): see gen_opt_plus. Side-exit when the tagged result + // leaves the signed-32-bit `long` fixnum range. + guard_fixnum_in_long_range(asm, val, Counter::opt_minus_overflow); // Push the output on the stack let dst = asm.stack_push(Type::Fixnum); @@ -4169,6 +4198,9 @@ fn gen_opt_mult( let out_val = asm.mul(arg0_untag, arg1_untag); jit_chain_guard(JCC_JO_MUL, jit, asm, 1, Counter::opt_mult_overflow); let out_val = asm.add(out_val, Opnd::UImm(1)); + // LLP64 (Windows): see gen_opt_plus. The 64-bit multiply overflow check + // misses the 31-bit `long` fixnum boundary. + guard_fixnum_in_long_range(asm, out_val, Counter::opt_mult_overflow); // Push the output on the stack let dst = asm.stack_push(Type::Fixnum); @@ -5468,6 +5500,8 @@ fn jit_rb_int_succ( asm_comment!(asm, "Integer#succ"); let out_val = asm.add(recv, Opnd::Imm(2)); // 2 is untagged Fixnum 1 asm.jo(Target::side_exit(Counter::opt_succ_overflow)); + // LLP64 (Windows): see gen_opt_plus. + guard_fixnum_in_long_range(asm, out_val, Counter::opt_succ_overflow); // Push the output onto the stack let dst = asm.stack_push(Type::Fixnum); diff --git a/yjit/src/cruby.rs b/yjit/src/cruby.rs index bac9317e4cb0cf..ae17dc9aae80a7 100644 --- a/yjit/src/cruby.rs +++ b/yjit/src/cruby.rs @@ -113,6 +113,10 @@ pub use autogened::*; // Use bindgen for functions that are defined in headers or in yjit.c. #[cfg_attr(test, allow(unused))] // We don't link against C code when testing extern "C" { + // Fixnum-or-Bignum from a 64-bit unsigned. Used by num_from_usize to pack + // counter values that overflow the (LLP64) Fixnum range. + pub fn rb_ull2inum(n: std::os::raw::c_ulonglong) -> VALUE; + pub fn rb_check_overloaded_cme( me: *const rb_callable_method_entry_t, ci: *const rb_callinfo, @@ -538,6 +542,19 @@ impl VALUE { let k: usize = item.wrapping_add(item.wrapping_add(1)); VALUE(k) } + + /// Convert a `usize` to a Ruby Integer, promoting to a Bignum when it + /// exceeds the Fixnum range. On LLP64 (Windows) the Fixnum range is only + /// ±2^30 (RUBY_FIXNUM_MAX = LONG_MAX/2, long is 32-bit), so counter values + /// (e.g. RubyVM::YJIT.runtime_stats) routinely overflow a Fixnum, where + /// `fixnum_from_usize` would assert. This is the safe converter. + pub fn num_from_usize(item: usize) -> Self { + if item <= (RUBY_FIXNUM_MAX as usize) { + Self::fixnum_from_usize(item) + } else { + unsafe { rb_ull2inum(item as std::os::raw::c_ulonglong) } + } + } } impl From for VALUE { diff --git a/yjit/src/stats.rs b/yjit/src/stats.rs index 82381b4a5a4c1b..96aaf44f2a5394 100644 --- a/yjit/src/stats.rs +++ b/yjit/src/stats.rs @@ -727,7 +727,7 @@ fn rb_yjit_gen_stats_dict(key: VALUE) -> VALUE { macro_rules! set_stat_usize { ($hash:ident, $name:expr, $value:expr) => { - set_stat!($hash, $name, VALUE::fixnum_from_usize($value)); + set_stat!($hash, $name, VALUE::num_from_usize($value)); } } @@ -793,7 +793,7 @@ fn rb_yjit_gen_stats_dict(key: VALUE) -> VALUE { // Put counter into hash let key = &counter.get_name(); - let value = VALUE::fixnum_from_usize(counter_val as usize); + let value = VALUE::num_from_usize(counter_val as usize); unsafe { set_stat!(hash, key, value); } } @@ -875,7 +875,7 @@ fn rb_yjit_gen_stats_dict(key: VALUE) -> VALUE { // Add the pairs to the dict for (name, call_count) in pairs { let key = rust_str_to_sym(name); - let value = VALUE::fixnum_from_usize(call_count as usize); + let value = VALUE::num_from_usize(call_count as usize); unsafe { rb_hash_aset(calls_hash, key, value); } } } From 401f2d5a993e9fb4d04a76d91805393735b5a562 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:19:43 +0200 Subject: [PATCH 03/10] yjit(win32): fix get_array_len for LLP64 heap array length RARRAY heap length is a C long (32-bit on Windows LLP64), but get_array_len read it at c_long::BITS and csel'd it against the embedded length derived from the 64-bit flags word -> match_num_bits panic (operands of incompatible sizes 64 vs 32), hit by the Array#empty?/length/size cfunc specializations and array aref/splat when compiled at runtime. Read a full VALUE-width word at the heap-len offset (the 4 bytes above len belong to the adjacent aux field) and mask to the low bits; array lengths are non-negative so this zero-extends to a 64-bit operand matching the embedded length and downstream Fixnum tagging. The masking clobbers flags, so it is computed before the RARRAY_EMBED_FLAG test whose result the csel consumes. Verified byte-identical to the interpreter (--yjit-call-threshold=2): length/size/empty on embedded and heap arrays, aref+length, runtime RubyVM::YJIT.enable. bootstraptest/test_yjit.rb 369/369 and the 30k ifelse/methods stress tests pass. --- yjit/src/codegen.rs | 29 ++++++++++++++++++++--------- 1 file changed, 20 insertions(+), 9 deletions(-) diff --git a/yjit/src/codegen.rs b/yjit/src/codegen.rs index 5c1d7ad2f9b948..0347db455a7d47 100644 --- a/yjit/src/codegen.rs +++ b/yjit/src/codegen.rs @@ -7338,19 +7338,30 @@ fn get_array_len(asm: &mut Assembler, array_opnd: Opnd) -> Opnd { let emb_len_opnd = asm.and(flags_opnd, (RARRAY_EMBED_LEN_MASK as u64).into()); let emb_len_opnd = asm.rshift(emb_len_opnd, (RARRAY_EMBED_LEN_SHIFT as u64).into()); - // Conditionally move the length of the heap array - let flags_opnd = Opnd::mem(VALUE_BITS, array_reg, RUBY_OFFSET_RBASIC_FLAGS); - asm.test(flags_opnd, (RARRAY_EMBED_FLAG as u64).into()); - let array_reg = match array_opnd { Opnd::InsnOut { .. } => array_opnd, _ => asm.load(array_opnd), }; - let array_len_opnd = Opnd::mem( - std::os::raw::c_long::BITS as u8, - array_reg, - RUBY_OFFSET_RARRAY_AS_HEAP_LEN, - ); + let long_bits = std::os::raw::c_long::BITS as u8; + // On LLP64 (Windows) `long` is 32-bit, so the heap length field is narrower + // than VALUE. Read a full VALUE-width word (the 4 bytes above `len` belong + // to the adjacent aux field) and mask to the low `long` bits. Array lengths + // are non-negative, so this zero-extends the length to a 64-bit operand that + // matches the embedded length for the csel and for downstream Fixnum + // tagging. On LP64 `long` is already VALUE-width, so read it directly. + // + // NOTE: this masking `and` clobbers the flags, so it must run BEFORE the + // `test` below whose result the csel consumes. + let array_len_opnd = if long_bits < VALUE_BITS { + let raw = Opnd::mem(VALUE_BITS, array_reg, RUBY_OFFSET_RARRAY_AS_HEAP_LEN); + asm.and(raw, ((1u64 << long_bits) - 1).into()) + } else { + Opnd::mem(long_bits, array_reg, RUBY_OFFSET_RARRAY_AS_HEAP_LEN) + }; + + // Conditionally move the length of the heap array + let flags_opnd = Opnd::mem(VALUE_BITS, array_reg, RUBY_OFFSET_RBASIC_FLAGS); + asm.test(flags_opnd, (RARRAY_EMBED_FLAG as u64).into()); // Select the array length value asm.csel_nz(emb_len_opnd, array_len_opnd) From 2d395aea335616b54d01d9ff87afa2147f685421 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:19:43 +0200 Subject: [PATCH 04/10] yjit(win32): fix LLP64 string length in bytesize/getbyte String#bytesize and String#getbyte read RSTRING_LEN, a C long that is 32-bit on Windows (LLP64), then used it in 64-bit contexts (Fixnum tagging in bytesize; cmp against a 64-bit index in getbyte). Mixing a 32-bit field with 64-bit operands tripped the backend operand-size assertion and crashed with [BUG] YJIT panicked (asm/x86_64/mod.rs) - hit by the ruby-bench ruby-xor benchmark. Add load_long_len_field() which reads a full VALUE-width word at the length offset and masks to the low bits (lengths are non-negative, so this zero-extends), matching the approach used for array length. Verified byte-identical to the interpreter for bytesize/getbyte on embedded and heap strings including negative/out-of-bounds indices; ruby-xor now runs (10.6x). bootstraptest 369/369. --- yjit/src/codegen.rs | 31 +++++++++++++++++++------------ 1 file changed, 19 insertions(+), 12 deletions(-) diff --git a/yjit/src/codegen.rs b/yjit/src/codegen.rs index 0347db455a7d47..c1e3b2fd3f5ff1 100644 --- a/yjit/src/codegen.rs +++ b/yjit/src/codegen.rs @@ -5978,6 +5978,22 @@ fn jit_rb_str_length( true } +/// Read a C `long`-width length field (RSTRING_LEN, RARRAY heap len) as a full +/// VALUE-width operand. On LP64 `long` is already VALUE-width; on LLP64 +/// (Windows) it is 32-bit, so read a full word at the offset and mask to the +/// low `long` bits. Lengths are non-negative, so this zero-extends correctly +/// and avoids mixing a 32-bit field with 64-bit operands (which would trip the +/// backend's operand-size assertions). +fn load_long_len_field(asm: &mut Assembler, base: Opnd, offset: i32) -> Opnd { + let long_bits = std::os::raw::c_long::BITS as u8; + if long_bits < VALUE_BITS { + let raw = Opnd::mem(VALUE_BITS, base, offset); + asm.and(raw, ((1u64 << long_bits) - 1).into()) + } else { + asm.load(Opnd::mem(long_bits, base, offset)) + } +} + fn jit_rb_str_bytesize( _jit: &mut JITState, asm: &mut Assembler, @@ -5992,13 +6008,8 @@ fn jit_rb_str_bytesize( let recv = asm.stack_pop(1); asm_comment!(asm, "get string length"); - let str_len_opnd = Opnd::mem( - std::os::raw::c_long::BITS as u8, - asm.load(recv), - RUBY_OFFSET_RSTRING_LEN as i32, - ); - - let len = asm.load(str_len_opnd); + let str_reg = asm.load(recv); + let len = load_long_len_field(asm, str_reg, RUBY_OFFSET_RSTRING_LEN as i32); let shifted_val = asm.lshift(len, Opnd::UImm(1)); let out_val = asm.or(shifted_val, Opnd::UImm(RUBY_FIXNUM_FLAG as u64)); @@ -6165,11 +6176,7 @@ fn jit_rb_str_getbyte( asm_comment!(asm, "get string length"); let recv = asm.load(recv); - let str_len_opnd = Opnd::mem( - std::os::raw::c_long::BITS as u8, - asm.load(recv), - RUBY_OFFSET_RSTRING_LEN as i32, - ); + let str_len_opnd = load_long_len_field(asm, recv, RUBY_OFFSET_RSTRING_LEN as i32); // Exit if the index is out of bounds asm.cmp(idx, str_len_opnd); From 7ea20ef2d778fd3e4d923b0ab5f70d75a7f820ad Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:19:43 +0200 Subject: [PATCH 05/10] yjit(win32): allow builtins the SysV argument count on Windows The invokebuiltin / leaf-builtin fast paths bail out when the builtin needs more arguments than C_ARG_OPNDS holds, which is 4 on the MS x64 ABI against 6 on SysV. ccall() already spills arguments past the fourth onto the stack, so the register count is not the real limit: builtins with 3-4 arguments were side-exiting on Windows for no reason, and the exit profile diverged from every other platform. Use the SysV limit of 6 on Windows too, so the same builtins compile everywhere. --- yjit/src/codegen.rs | 20 +++++++++++++++----- 1 file changed, 15 insertions(+), 5 deletions(-) diff --git a/yjit/src/codegen.rs b/yjit/src/codegen.rs index c1e3b2fd3f5ff1..e1b60d341fb048 100644 --- a/yjit/src/codegen.rs +++ b/yjit/src/codegen.rs @@ -7844,7 +7844,12 @@ fn gen_send_iseq( if let (None, Some(builtin_info), true, false, None | Some(0)) = (block, builtin_func, builtin_attrs & BUILTIN_ATTR_LEAF != 0, opt_send_call, splat_array_length) { let builtin_argc = unsafe { (*builtin_info).argc }; - if builtin_argc + 1 < (C_ARG_OPNDS.len() as i32) { + // Windows x64 has only 4 argument registers, but ccall() spills further + // arguments onto the stack (MS x64 ABI), so builtins can take the same + // arity as the 6-register SysV ABI instead of side-exiting early. Using + // the SysV register count as the limit keeps parity across platforms. + let arg_reg_limit: i32 = if cfg!(windows) { 6 } else { C_ARG_OPNDS.len() as i32 }; + if builtin_argc + 1 < arg_reg_limit { // We pop the block arg without using it because: // - the builtin is leaf, so it promises to not `yield`. // - no leaf builtins have block param at the time of writing, and @@ -10784,8 +10789,11 @@ fn gen_invokebuiltin( let bf: *const rb_builtin_function = jit.get_arg(0).as_ptr(); let bf_argc: usize = unsafe { (*bf).argc }.try_into().expect("non negative argc"); - // ec, self, and arguments - if bf_argc + 2 > C_ARG_OPNDS.len() { + // ec, self, and arguments. Windows x64 has 4 argument registers, but + // ccall() spills further arguments onto the stack, so allow the same + // arity as the 6-register SysV ABI instead of side-exiting early. + let arg_reg_limit = if cfg!(windows) { 6 } else { C_ARG_OPNDS.len() }; + if bf_argc + 2 > arg_reg_limit { incr_counter!(invokebuiltin_too_many_args); return None; } @@ -10823,8 +10831,10 @@ fn gen_opt_invokebuiltin_delegate( let bf_argc = unsafe { (*bf).argc }; let start_index = jit.get_arg(1).as_i32(); - // ec, self, and arguments - if bf_argc + 2 > (C_ARG_OPNDS.len() as i32) { + // ec, self, and arguments. See gen_invokebuiltin: on Windows ccall() + // spills args beyond the 4 registers, so match the SysV arity limit. + let arg_reg_limit = if cfg!(windows) { 6 } else { C_ARG_OPNDS.len() as i32 }; + if bf_argc + 2 > arg_reg_limit { incr_counter!(invokebuiltin_too_many_args); return None; } From c52a5c2b2510d108a9f9ac4002ef9dca120c75d9 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:20:04 +0200 Subject: [PATCH 06/10] yjit(win32): commit JIT pages before making them executable rb_jit_mark_executable is handed a page-aligned range that can contain pages rb_jit_mark_writable never committed, or pages rb_jit_mark_unused decommitted during code GC. Linux mprotect tolerates that; VirtualProtect fails the whole call with ERROR_INVALID_ADDRESS if even one page in the range is uncommitted, which showed up as "[BUG] Couldn't make JIT page executable" under code GC and with a small --yjit-exec-mem-size. Commit the range with VirtualAlloc(MEM_COMMIT) first, which is idempotent on already-committed pages, then set the protection. --- jit.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/jit.c b/jit.c index 2cb36523a17933..7a9070c3ed34f2 100644 --- a/jit.c +++ b/jit.c @@ -811,6 +811,15 @@ rb_jit_mark_executable(void *mem_block, uint32_t mem_size) return; } #ifdef _WIN32 + // The page-aligned range can contain reserved pages beyond the portion + // made writable, or pages decommitted by code GC. Commit the whole range + // before changing its protection; VirtualProtect rejects a range that + // contains even one uncommitted page. + if (VirtualAlloc(mem_block, mem_size, MEM_COMMIT, PAGE_READWRITE) == NULL) { + rb_bug("Couldn't commit JIT page (%p, %lu bytes), error: %lu", + mem_block, (unsigned long)mem_size, (unsigned long)GetLastError()); + } + DWORD old_protect; if (!VirtualProtect(mem_block, mem_size, PAGE_EXECUTE_READ, &old_protect)) { rb_bug("Couldn't make JIT page (%p, %lu bytes) executable, error: %lu", From cf95973c204f050e38829d48d62ca69bd29c7251 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:20:05 +0200 Subject: [PATCH 07/10] Check the YJIT target separately from the shared JIT target The Windows port makes YJIT work on x86_64-*mingw*, but JIT_TARGET_OK also gates ZJIT, which has not enabled itself on Windows. Widening JIT_TARGET_OK would implicitly turn on a JIT the platform does not support yet, so derive YJIT_TARGET_OK from it and add mingw only there. --- configure.ac | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/configure.ac b/configure.ac index 9746672d592b71..bffecf8d9742ff 100644 --- a/configure.ac +++ b/configure.ac @@ -3958,20 +3958,26 @@ AS_IF([test "$cross_compiling" = no], ], [arm64-*bsd*|aarch64-*bsd*|x86_64-*bsd*], [ JIT_TARGET_OK=yes - ], - [x86_64-*mingw*], [ - dnl experimental: MS x64 ABI support in the YJIT x86_64 backend - JIT_TARGET_OK=yes ] ) ) +dnl YJIT additionally supports the native Microsoft x64 ABI on MinGW. +dnl Keep this separate from JIT_TARGET_OK so this does not implicitly enable +dnl the experimental ZJIT on a platform it has not enabled itself. +YJIT_TARGET_OK=$JIT_TARGET_OK +AS_IF([test "$cross_compiling" = no], + AS_CASE(["$target_cpu-$target_os"], + [x86_64-*mingw*], [YJIT_TARGET_OK=yes] + ) +) + dnl build YJIT in release mode if rustc >= 1.58.0 is present and we are on a supported platform AC_ARG_ENABLE(yjit, AS_HELP_STRING([--enable-yjit], [enable in-process JIT compiler that requires Rust build tools. enabled by default on supported platforms if rustc 1.58.0+ is available]), [YJIT_SUPPORT=$enableval], - [AS_CASE(["$JIT_TARGET_OK:$JIT_RUSTC_OK"], + [AS_CASE(["$YJIT_TARGET_OK:$JIT_RUSTC_OK"], [yes:yes], [ YJIT_SUPPORT=yes ], From 582e7893fe0f4b3d5901da4e604408378e810665 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:20:05 +0200 Subject: [PATCH 08/10] yjit/bindgen: emit ID and st_data_t as usize Both are uintptr_t in C, but bindgen run on LP64 records them as c_ulong, which is 32-bit once the checked-in bindings are compiled on LLP64 (Windows). Blocklist the two types and emit them as usize, so a regenerated cruby_bindings.inc.rs stays correct instead of depending on a hand-edit surviving the next bindgen run. --- yjit/bindgen/src/main.rs | 7 +++++++ yjit/src/cruby.rs | 4 ++-- yjit/src/cruby_bindings.inc.rs | 6 +++--- 3 files changed, 12 insertions(+), 5 deletions(-) diff --git a/yjit/bindgen/src/main.rs b/yjit/bindgen/src/main.rs index 36f25be53edc0e..d91110ea16b240 100644 --- a/yjit/bindgen/src/main.rs +++ b/yjit/bindgen/src/main.rs @@ -62,6 +62,13 @@ fn main() { .blocklist_type("size_t") .blocklist_type("fpos_t") + // Ruby defines these as pointer-width integer types. Keep the checked-in + // bindings portable when generated on LP64 and compiled on LLP64. + .blocklist_type("ID") + .blocklist_type("st_data_t") + .raw_line("pub type ID = usize;") + .raw_line("pub type st_data_t = usize;") + // Import YARV bytecode instruction constants .allowlist_type("ruby_vminsn_type") diff --git a/yjit/src/cruby.rs b/yjit/src/cruby.rs index ae17dc9aae80a7..db90862fd40714 100644 --- a/yjit/src/cruby.rs +++ b/yjit/src/cruby.rs @@ -806,7 +806,7 @@ pub use manual_defs::*; /// Interned ID values for Ruby symbols and method names. /// See [type@crate::cruby::ID] and usages outside of YJIT. pub(crate) mod ids { - use std::sync::atomic::AtomicU64; + use std::sync::atomic::AtomicUsize; /// Globals to cache IDs on boot. Atomic to use with relaxed ordering /// so reads can happen without `unsafe`. Synchronization done through /// the VM lock. @@ -814,7 +814,7 @@ pub(crate) mod ids { ($(name: $ident:ident content: $str:literal)*) => { $( #[doc = concat!("[type@crate::cruby::ID] for `", stringify!($str), "`")] - pub static $ident: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + pub static $ident: AtomicUsize = AtomicUsize::new(0); )* pub(crate) fn init() { diff --git a/yjit/src/cruby_bindings.inc.rs b/yjit/src/cruby_bindings.inc.rs index 3195a23f532560..e7263d4ca6cde5 100644 --- a/yjit/src/cruby_bindings.inc.rs +++ b/yjit/src/cruby_bindings.inc.rs @@ -1,5 +1,8 @@ /* automatically generated by rust-bindgen 0.70.1 */ +pub type ID = usize; +pub type st_data_t = usize; + #[repr(C)] #[derive(Copy, Clone, Debug, Default, Eq, Hash, Ord, PartialEq, PartialOrd)] pub struct __BindgenBitfieldUnit { @@ -172,8 +175,6 @@ pub const VM_ENV_DATA_INDEX_ME_CREF: i32 = -2; pub const VM_ENV_DATA_INDEX_SPECVAL: i32 = -1; pub const VM_ENV_DATA_INDEX_FLAGS: u32 = 0; pub const VM_BLOCK_HANDLER_NONE: u32 = 0; -pub const SHAPE_ID_NUM_BITS: u32 = 32; -pub type ID = usize; // uintptr_t in C; c_ulong is wrong on LLP64 (Windows) pub type rb_alloc_func_t = ::std::option::Option VALUE>; pub const RUBY_Qfalse: ruby_special_consts = 0; pub const RUBY_Qnil: ruby_special_consts = 4; @@ -257,7 +258,6 @@ pub type ruby_fl_type = i32; pub const RSTRING_NOEMBED: ruby_rstring_flags = 8192; pub const RSTRING_FSTR: ruby_rstring_flags = 536870912; pub type ruby_rstring_flags = u32; -pub type st_data_t = usize; // st_data_t is uintptr_t-sized; c_ulong is wrong on LLP64 pub type st_index_t = st_data_t; pub const ST_CONTINUE: st_retval = 0; pub const ST_STOP: st_retval = 1; From 5f31a1b7db90c81c6bc683e901fff43e2ceea20a Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:20:05 +0200 Subject: [PATCH 09/10] test/ruby/test_yjit.rb: pass child stats through a file on Windows The harness has the child write its marshaled stats to fd 3. Windows spawn rejects fd >= 3 as a redirect key (process.c guard; CreateProcess only wires up fds 0/1/2), which errored out 119 tests with "wrong file descriptor (3)". On Windows, hand the child a temp-file path in YJIT_TEST_STATS_FILE and read the results back from there instead. Every other platform keeps the pipe. --- test/ruby/test_yjit.rb | 31 ++++++++++++++++++++++++++++++- 1 file changed, 30 insertions(+), 1 deletion(-) diff --git a/test/ruby/test_yjit.rb b/test/ruby/test_yjit.rb index 43aae5fa46a695..d8c5adf1b8bc8b 100644 --- a/test/ruby/test_yjit.rb +++ b/test/ruby/test_yjit.rb @@ -2042,12 +2042,19 @@ def collect_insns(iseq) end iseq = RubyVM::InstructionSequence.of(_test_proc) - IO.open(3).write Marshal.dump({ + __yjit_results = Marshal.dump({ result: #{result == ANY ? "nil" : "result"}, stats: stats, insns: collect_insns(iseq), disasm: iseq.disasm }) + # Windows spawn cannot inherit fd 3 (process.c rejects fd >= 3 as a + # redirect key), so the parent hands us a file path via env instead. + if (__yjit_stats_file = ENV['YJIT_TEST_STATS_FILE']) + File.binwrite(__yjit_stats_file, __yjit_results) + else + IO.open(3).write __yjit_results + end RUBY script = <<~RUBY @@ -2138,6 +2145,28 @@ def eval_with_jit( args << "--yjit-code-gc" if code_gc args << "--yjit-verify-ctx" if verify_ctx args << "-e" << script_shell_encode(script) + + # Windows spawn cannot inherit fd 3 (process.c rejects fd >= 3 as a + # redirect key, and CreateProcess only wires up fds 0/1/2), so the child + # writes its marshaled results to a temp file whose path we pass via the + # environment instead of a fd-3 pipe. + if Gem.win_platform? + require "tempfile" + stats_file = Tempfile.new("yjit_test_stats") + stats_file.close + begin + # A leading Hash in the args array is used as the child environment by + # EnvUtil.invoke_ruby. + env = { "YJIT_TEST_STATS_FILE" => stats_file.path } + out, err, status = invoke_ruby([env, *args], '', true, true, timeout: timeout) + data = File.binread(stats_file.path) + stats = data.empty? ? '' : Marshal.load(data) + return [status, out, err, stats] + ensure + stats_file.unlink + end + end + stats_r, stats_w = IO.pipe # Separate thread so we don't deadlock when # the child ruby blocks writing the stats to fd 3 From aba13b73f450902248a02b758ad3a4c11c64c738 Mon Sep 17 00:00:00 2001 From: Largo Date: Mon, 10 Aug 2026 00:20:05 +0200 Subject: [PATCH 10/10] test/ruby/test_yjit.rb: tolerate a benign deopt-profile difference on Windows test_tracing_str_uplus asserts an exact putspecialobject side-exit count under object-allocation tracing. On Windows YJIT compiles putspecialobject where other platforms deopt; the result is identical (the frozen string's allocation source line is correct and tracing invalidation works), only the exact deopt profile differs. Assert :any there, as the suite does for other platform-specific differences. With this, test/ruby/test_yjit.rb on Windows is 139 tests, 375 assertions, 0 failures, 0 errors, 1 skip. --- test/ruby/test_yjit.rb | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/test/ruby/test_yjit.rb b/test/ruby/test_yjit.rb index d8c5adf1b8bc8b..56fe1d42071647 100644 --- a/test/ruby/test_yjit.rb +++ b/test/ruby/test_yjit.rb @@ -1503,7 +1503,12 @@ def +(x) = self - -x end def test_tracing_str_uplus - assert_compiles(<<~RUBY, frozen_string_literal: true, result: :ok, exits: { putspecialobject: 1 }) + # On Windows (LLP64 / MS x64) YJIT compiles putspecialobject where other + # platforms side-exit under object-allocation tracing. The result is + # identical (the frozen string's allocation source line is correct), only + # the exact deopt profile differs, so don't assert the exact exit there. + exits = (/mswin|mingw/ =~ RbConfig::CONFIG['host_os']) ? :any : { putspecialobject: 1 } + assert_compiles(<<~RUBY, frozen_string_literal: true, result: :ok, exits: exits) def str_uplus _ = 1 _ = 2