// Bounds check elimination. // // We perform BCE on two kinds of address expressions: on constant heap pointers // that are known to be in the heap or will be handled by the out-of-bounds trap // handler; and on local variables that have been checked in dominating code // without being updated since. // // For an access through a constant heap pointer + an offset we can eliminate // the bounds check if the sum of the address and offset is below the sum of the // minimum memory length and the offset guard length. // // For an access through a local variable + an offset we can eliminate the // bounds check if the local variable has already been checked and has not been // updated since, and the offset is less than the guard limit. // // To track locals for which we can eliminate checks we use a bit vector // bceSafe_ that has a bit set for those locals whose bounds have been checked // and which have not subsequently been set. Initially this vector is zero. // // In straight-line code a bit is set when we perform a bounds check on an // access via the local and is reset when the variable is updated. // // In control flow, the bit vector is manipulated as follows. Each ControlItem // has a value bceSafeOnEntry, which is the value of bceSafe_ on entry to the // item, and a value bceSafeOnExit, which is initially ~0. On a branch (br, // brIf, brTable), we always AND the branch target's bceSafeOnExit with the // value of bceSafe_ at the branch point. On exiting an item by falling out of // it, provided we're not in dead code, we AND the current value of bceSafe_ // into the item's bceSafeOnExit. Additional processing depends on the item // type: // // - After a block, set bceSafe_ to the block's bceSafeOnExit. // // - On loop entry, after pushing the ControlItem, set bceSafe_ to zero; the // back edges would otherwise require us to iterate to a fixedpoint. // // - After a loop, the bceSafe_ is left unchanged, because only fallthrough // control flow will reach that point and the bceSafe_ value represents the // correct state of the fallthrough path. // // - Set bceSafe_ to the ControlItem's bceSafeOnEntry at both the 'then' branch // and the 'else' branch. // // - After an if-then-else, set bceSafe_ to the if-then-else's bceSafeOnExit. // // - After an if-then, set bceSafe_ to the if-then's bceSafeOnExit AND'ed with // the if-then's bceSafeOnEntry. // // Finally, when the debugger allows locals to be mutated we must disable BCE // for references via a local, by returning immediately from bceCheckLocal if // compilerEnv_.debugEnabled() is true.
void BaseCompiler::bceCheckLocal(MemoryAccessDesc* access, AccessCheck* check,
uint32_t local) { // We only eliminate bounds checks for memory 0 if (access->memoryIndex() != 0) { return;
}
if (local >= sizeof(BCESet) * 8) { return;
}
#ifdef ENABLE_WASM_CUSTOM_PAGE_SIZES if (codeMeta_.memories[0].pageSize() != PageSize::Standard) { // We do not have guard pages, so we cannot perform this optimization. return;
} #endif
// Alignment check elimination. // // Alignment checks for atomic operations can be omitted if the pointer is a // constant and the pointer + offset is aligned. Alignment checking that can't // be omitted can still be simplified by checking only the pointer if the offset // is aligned. // // (In addition, alignment checking of the pointer can be omitted if the pointer // has been checked in dominating code, but we don't do that yet.)
// Validation ensures that the offset is in 32-bit range, and the calculation // of the limit cannot overflow due to our choice of HugeOffsetGuardLimit. #ifdef WASM_SUPPORTS_HUGE_MEMORY
static_assert(MaxMemory32StandardPagesValidation * StandardPageSizeBytes <=
UINT64_MAX - HugeOffsetGuardLimit); #endif
uint64_t ea = uint64_t(addr) + uint64_t(access->offset32());
uint64_t finalAddress = ea + access->byteSize();
uint64_t limit = codeMeta_.memories[access->memoryIndex()].initialLength() +
offsetGuardLimit;
void BaseCompiler::boundsCheck4GBOrLargerAccess(uint32_t memoryIndex, unsigned byteSize,
RegPtr instance, RegI32 ptr,
Label* ok) { #ifdef JS_64BIT // Extend the value to 64 bits, check the 64-bit value against the 64-bit // bound, then chop back to 32 bits. On most platform the extending and // chopping are no-ops. It's important that the value we end up with has // flowed through the Spectre mask
// Note, ptr and ptr64 are the same register.
RegI64 ptr64 = fromI32(ptr);
// In principle there may be non-zero bits in the upper bits of the // register; clear them. # ifdef RABALDR_ZERO_EXTENDS
masm.debugAssertCanonicalInt32(ptr); # else
masm.move32To64ZeroExtend(ptr, ptr64); # endif
// Restore the value to the canonical form for a 32-bit value in a // 64-bit register and/or the appropriate form for further use in the // indexing instruction. # ifdef RABALDR_ZERO_EXTENDS // The canonical value is zero-extended; we already have that. # else
masm.move64To32(ptr64, ptr); # endif #else // No support needed, we have max 2GB heap on 32-bit
MOZ_CRASH("No 32-bit support"); #endif
}
void BaseCompiler::boundsCheckBelow4GBAccess(uint32_t memoryIndex, unsigned byteSize, RegPtr instance,
RegI32 ptr, Label* ok) { // If the memory's max size is known to be smaller than 64K pages exactly, // we can use a 32-bit check and avoid extension and wrapping.
masm.wasmBoundsCheck32(Assembler::Below, ptr,
Address(instance, instanceOffsetOfBoundsCheckLimit(
memoryIndex, byteSize)),
ok);
}
void BaseCompiler::boundsCheck4GBOrLargerAccess(uint32_t memoryIndex, unsigned byteSize,
RegPtr instance, RegI64 ptr,
Label* ok) { // Any Spectre mitigation will appear to update the ptr64 register.
masm.wasmBoundsCheck64(Assembler::Below, ptr,
Address(instance, instanceOffsetOfBoundsCheckLimit(
memoryIndex, byteSize)),
ok);
}
void BaseCompiler::boundsCheckBelow4GBAccess(uint32_t memoryIndex, unsigned byteSize, RegPtr instance,
RegI64 ptr, Label* ok) { // The bounds check limit is valid to 64 bits, so there's no sense in doing // anything complicated here. There may be optimization paths here in the // future and they may differ on 32-bit and 64-bit.
boundsCheck4GBOrLargerAccess(memoryIndex, byteSize, instance, ptr, ok);
}
// Make sure the ptr could be used as an index register. staticinlinevoid ToValidIndex(MacroAssembler& masm, RegI32 ptr) { #ifdefined(JS_CODEGEN_MIPS64) || defined(JS_CODEGEN_LOONG64) || \ defined(JS_CODEGEN_RISCV64) // When ptr is used as an index, it will be added to a 64-bit register. // So we should explicitly promote ptr to 64-bit. Since now ptr holds a // unsigned 32-bit value, we zero-extend it to 64-bit here.
masm.move32To64ZeroExtend(ptr, Register64(ptr)); #endif
}
// Fold offset if necessary for further computations. if (access->offset64() >= offsetGuardLimit ||
access->offset64() > UINT32_MAX ||
(access->isAtomic() && !check->omitAlignmentCheck &&
!check->onlyPointerAlignment)) {
Label ok;
branchAddNoOverflow(access->offset64(), ptr, &ok);
trap(Trap::OutOfBounds);
masm.bind(&ok);
access->clearOffset();
check->onlyPointerAlignment = true;
}
// Alignment check if required.
if (access->isAtomic() && !check->omitAlignmentCheck) {
MOZ_ASSERT(check->onlyPointerAlignment); // We only care about the low pointer bits here.
Label ok;
branchTestLowZero(ptr, Imm32(access->byteSize() - 1), &ok);
trap(Trap::UnalignedAccess);
masm.bind(&ok);
}
// Ensure no instance if we don't need it.
if (codeMeta_.hugeMemoryEnabled(access->memoryIndex()) &&
access->memoryIndex() == 0) { // We have HeapReg and no bounds checking and need load neither // memoryBase nor boundsCheckLimit from instance.
MOZ_ASSERT_IF(check->omitBoundsCheck, instance.isInvalid());
} #ifdef WASM_HAS_HEAPREG // We have HeapReg and don't need to load the memoryBase from instance.
MOZ_ASSERT_IF(check->omitBoundsCheck && access->memoryIndex() == 0,
instance.isInvalid()); #endif
if (!codeMeta_.hugeMemoryEnabled(access->memoryIndex()) &&
!check->omitBoundsCheck) {
Label ok; #ifdef JS_64BIT // The checking depends on how many bits are in the pointer and how many // bits are in the bound. if (!codeMeta_.memories[access->memoryIndex()]
.boundsCheckLimitIsAlways32Bits() &&
MaxMemoryBytes(codeMeta_.memories[access->memoryIndex()].addressType(),
codeMeta_.memories[access->memoryIndex()].pageSize()) >= 0x100000000) {
boundsCheck4GBOrLargerAccess(access->memoryIndex(), access->byteSize(),
instance, ptr, &ok);
} else {
boundsCheckBelow4GBAccess(access->memoryIndex(), access->byteSize(),
instance, ptr, &ok);
} #else
boundsCheckBelow4GBAccess(access->memoryIndex(), access->byteSize(),
instance, ptr, &ok); #endif
trap(Trap::OutOfBounds);
masm.bind(&ok);
}
RegPtr BaseCompiler::maybeLoadMemoryBaseForAccess(
RegPtr instance, const MemoryAccessDesc* access) { #ifdef JS_CODEGEN_X86 // x86 adds the memory base to the wasm pointer directly using an addressing // mode and doesn't require the memory base to be loaded to a register. return RegPtr(); #endif
bool BaseCompiler::needInstanceForAccess(const MemoryAccessDesc* access, const AccessCheck& check) { #ifndef WASM_HAS_HEAPREG // Platform requires instance for memory base. returntrue; #else if (access->memoryIndex() != 0) { // Need instance to load the memory base returntrue;
} return !codeMeta_.hugeMemoryEnabled(access->memoryIndex()) &&
!check.omitBoundsCheck; #endif
}
RegPtr BaseCompiler::maybeLoadInstanceForAccess(const MemoryAccessDesc* access, const AccessCheck& check) { if (needInstanceForAccess(access, check)) { #ifdef RABALDR_PIN_INSTANCE // NOTE, returning InstanceReg here depends for correctness on *ALL* // clients not attempting to free this register and not push it on the value // stack. // // We have assertions in place to guard against that, so the risk of the // leaky abstraction is acceptable. performRegisterLeakCheck() will ensure // that after every bytecode, the union of available registers from the // regalloc and used registers from the stack equals the set of allocatable // registers at startup. Thus if the instance is freed incorrectly it will // end up in that union via the regalloc, and if it is pushed incorrectly it // will end up in the union via the stack. return RegPtr(InstanceReg); #else
RegPtr instance = need<RegPtr>();
fr.loadInstancePtr(instance); return instance; #endif
} return RegPtr::Invalid();
}
////////////////////////////////////////////////////////////////////////////// // // Load and store.
void BaseCompiler::executeLoad(MemoryAccessDesc* access, AccessCheck* check,
RegPtr instance, RegPtr memoryBase, RegI32 ptr,
AnyReg dest, RegI32 temp) { // Emit the load. At this point, 64-bit offsets will have been folded away by // prepareMemoryAccess. #ifdefined(JS_CODEGEN_X64)
MOZ_ASSERT(temp.isInvalid());
Operand srcAddr(memoryBase, ptr, TimesOne, access->offset32());
if (dest.tag == AnyReg::I64) {
MOZ_ASSERT(dest.i64() == specific_.abiReturnRegI64);
masm.wasmLoadI64(*access, srcAddr, dest.i64());
} else { // For 8 bit loads, this will generate movsbl or movzbl, so // there's no constraint on what the output register may be.
masm.wasmLoad(*access, srcAddr, dest.any());
} #elifdefined(JS_CODEGEN_MIPS64) if (IsUnaligned(*access)) { switch (dest.tag) { case AnyReg::I64:
masm.wasmUnalignedLoadI64(*access, memoryBase, ptr, ptr, dest.i64(),
temp); break; case AnyReg::F32:
masm.wasmUnalignedLoadFP(*access, memoryBase, ptr, ptr, dest.f32(),
temp); break; case AnyReg::F64:
masm.wasmUnalignedLoadFP(*access, memoryBase, ptr, ptr, dest.f64(),
temp); break; case AnyReg::I32:
masm.wasmUnalignedLoad(*access, memoryBase, ptr, ptr, dest.i32(), temp); break; default:
MOZ_CRASH("Unexpected type");
}
} else { if (dest.tag == AnyReg::I64) {
masm.wasmLoadI64(*access, memoryBase, ptr, ptr, dest.i64());
} else {
masm.wasmLoad(*access, memoryBase, ptr, ptr, dest.any());
}
} #elifdefined(JS_CODEGEN_ARM)
MOZ_ASSERT(temp.isInvalid()); if (dest.tag == AnyReg::I64) {
masm.wasmLoadI64(*access, memoryBase, ptr, ptr, dest.i64());
} else {
masm.wasmLoad(*access, memoryBase, ptr, ptr, dest.any());
} #elifdefined(JS_CODEGEN_ARM64)
MOZ_ASSERT(temp.isInvalid()); if (dest.tag == AnyReg::I64) {
masm.wasmLoadI64(*access, memoryBase, ptr, dest.i64());
} else {
masm.wasmLoad(*access, memoryBase, ptr, dest.any());
} #elifdefined(JS_CODEGEN_LOONG64)
MOZ_ASSERT(temp.isInvalid()); if (dest.tag == AnyReg::I64) {
masm.wasmLoadI64(*access, memoryBase, ptr, ptr, dest.i64());
} else {
masm.wasmLoad(*access, memoryBase, ptr, ptr, dest.any());
} #elifdefined(JS_CODEGEN_RISCV64)
MOZ_ASSERT(temp.isInvalid()); if (dest.tag == AnyReg::I64) {
masm.wasmLoadI64(*access, memoryBase, ptr, dest.i64());
} else {
masm.wasmLoad(*access, memoryBase, ptr, dest.any());
} #else
MOZ_CRASH("BaseCompiler platform hook: load"); #endif
}
// ptr and dest may be the same iff dest is I32. // This may destroy ptr even if ptr and dest are not the same. void BaseCompiler::load(MemoryAccessDesc* access, AccessCheck* check,
RegPtr instance, RegPtr memoryBase, RegI32 ptr,
AnyReg dest, RegI32 temp) {
prepareMemoryAccess(access, check, instance, ptr);
executeLoad(access, check, instance, memoryBase, ptr, dest, temp);
}
#if !defined(JS_64BIT) // On 32-bit systems we have a maximum 2GB heap and bounds checking has // been applied to ensure that the 64-bit pointer is valid. return executeLoad(access, check, instance, memoryBase, RegI32(ptr.low), dest,
maybeFromI64(temp)); #elifdefined(JS_CODEGEN_X64) || defined(JS_CODEGEN_ARM64) // On x64 and arm64 the 32-bit code simply assumes that the high bits of the // 64-bit pointer register are zero and performs a 64-bit add. Thus the code // generated is the same for the 64-bit and the 32-bit case. return executeLoad(access, check, instance, memoryBase, RegI32(ptr.reg), dest,
maybeFromI64(temp)); #elifdefined(JS_CODEGEN_MIPS64) || defined(JS_CODEGEN_LOONG64) // On mips64 and loongarch64, the 'prepareMemoryAccess' function will make // sure that ptr holds a valid 64-bit index value. Thus the code generated in // 'executeLoad' is the same for the 64-bit and the 32-bit case. return executeLoad(access, check, instance, memoryBase, RegI32(ptr.reg), dest,
maybeFromI64(temp)); #elifdefined(JS_CODEGEN_RISCV64) // RISCV the 'prepareMemoryAccess' function will make // sure that ptr holds a valid 64-bit index value. Thus the code generated in // 'executeLoad' is the same for the 64-bit and the 32-bit case. return executeLoad(access, check, instance, memoryBase, RegI32(ptr.reg), dest,
maybeFromI64(temp)); #else
MOZ_CRASH("Missing platform hook"); #endif
}
void BaseCompiler::executeStore(MemoryAccessDesc* access, AccessCheck* check,
RegPtr instance, RegPtr memoryBase, RegI32 ptr,
AnyReg src, RegI32 temp) { // Emit the store. At this point, 64-bit offsets will have been folded away by // prepareMemoryAccess. #ifdefined(JS_CODEGEN_X64)
MOZ_ASSERT(temp.isInvalid());
Operand dstAddr(memoryBase, ptr, TimesOne, access->offset32());
// ptr and src must not be the same register. // This may destroy ptr and src. void BaseCompiler::store(MemoryAccessDesc* access, AccessCheck* check,
RegPtr instance, RegPtr memoryBase, RegI32 ptr,
AnyReg src, RegI32 temp) {
prepareMemoryAccess(access, check, instance, ptr);
executeStore(access, check, instance, memoryBase, ptr, src, temp);
}
////////////////////////////////////////////////////////////////////////////// // // Atomic operations. // // The atomic operations have very diverse per-platform needs for register // allocation and temps. To handle that, the implementations are structured as // a per-operation framework method that calls into platform-specific helpers // (usually called PopAndAllocate, Perform, and Deallocate) in a per-operation // namespace. This structure results in a little duplication and boilerplate // but is otherwise clean and flexible and keeps code and supporting definitions // entirely co-located.
// Some consumers depend on the returned Address not incorporating instance, as // instance may be the scratch register. // // RegAddressType is RegI32 for Memory32 and RegI64 for Memory64. template <typename RegAddressType>
Address BaseCompiler::prepareAtomicMemoryAccess(MemoryAccessDesc* access,
AccessCheck* check,
RegPtr instance,
RegAddressType ptr) {
MOZ_ASSERT(needInstanceForAccess(access, *check) == instance.isValid());
prepareMemoryAccess(access, check, instance, ptr);
staticvoid Allocate(BaseCompiler* bc, RegI64* rd, RegI64* temp) { // The result is in edx:eax, and we need ecx:ebx as a temp. But ebx will also // be used as a scratch, so don't manage that here.
bc->needI32(bc->specific_.ecx);
*temp = bc->specific_.ecx_ebx;
bc->needI64(bc->specific_.edx_eax);
*rd = bc->specific_.edx_eax;
}
struct Temps { // On x86 we use the ScratchI32 for the temp, otherwise we'd run out of // registers for 64-bit operations. # ifdefined(JS_CODEGEN_X64)
RegI32 t0; # endif
};
staticvoid PopAndAllocate(BaseCompiler* bc, ValType type,
Scalar::Type viewType, AtomicOp op, RegI32* rd,
RegI32* rv, Temps* temps) {
bc->needI32(bc->specific_.eax); if (op == AtomicOp::Add || op == AtomicOp::Sub) { // We use xadd, so source and destination are the same. Using // eax here is overconstraining, but for byte operations on x86 // we do need something with a byte register. if (type == ValType::I64) {
*rv = bc->popI64ToSpecificI32(bc->specific_.eax);
} else {
*rv = bc->popI32ToSpecific(bc->specific_.eax);
}
*rd = *rv;
} else { // We use a cmpxchg loop. The output must be eax; the input // must be in a separate register since it may be used several // times. if (type == ValType::I64) {
*rv = bc->popI64ToI32();
} else {
*rv = bc->popI32();
}
*rd = bc->specific_.eax; # ifdef JS_CODEGEN_X64
temps->t0 = bc->needI32(); # endif
}
}
staticvoid PopAndAllocate(BaseCompiler* bc, AtomicOp op, RegI64* rd,
RegI64* rv, RegI64* temp) { if (op == AtomicOp::Add || op == AtomicOp::Sub) { // We use xaddq, so input and output must be the same register.
*rv = bc->popI64();
*rd = *rv;
} else { // We use a cmpxchgq loop, so the output must be rax and we need a temp.
bc->needI64(bc->specific_.rax);
*rd = bc->specific_.rax;
*rv = bc->popI64();
*temp = bc->needI64();
}
}
// Register allocation is tricky, see comments at atomic_xchg64 below. // // - Initially rv=ecx:edx and eax is reserved, rd=unallocated. // - Then rp is popped into esi+edi because those are the only available. // - The Setup operation makes rd=edx:eax. // - Deallocation then frees only the ecx part of rv. // // The temp is unused here.
staticvoid PopAndAllocate(BaseCompiler* bc, AtomicOp op, RegI64* rd,
RegI64* rv, RegI64* temp) { // We use a ldrex/strexd loop so the temp and the output must be // odd/even pairs.
*rv = bc->popI64();
*temp = bc->needI64Pair();
*rd = bc->needI64Pair();
}
// Register allocation is tricky in several ways. // // - For a 64-bit access on memory64 we need six registers for rd, rv, and rp, // but have only five (as the temp ebx is needed too), so we target all // registers explicitly to make sure there's space. // // - We'll be using cmpxchg8b, and when we do the operation, rv must be in // ecx:ebx, and rd must be edx:eax. We can't use ebx for rv initially because // we need ebx for a scratch also, so use a separate temp and move the value // to ebx just before the operation. // // In sum: // // - Initially rv=ecx:edx and eax is reserved, rd=unallocated. // - Then rp is popped into esi+edi because those are the only available. // - The Setup operation makes rv=ecx:ebx and rd=edx:eax and moves edx->ebx. // - Deallocation then frees only the ecx part of rv.
template <typename RegAddressType> staticvoid PopAndAllocate(BaseCompiler* bc, RegI64* rexpect, RegI64* rnew,
RegI64* rd) { // For cmpxchg, the expected value and the result are both in rax.
bc->needI64(bc->specific_.rax);
*rnew = bc->popI64();
*rexpect = bc->popI64ToSpecific(bc->specific_.rax);
*rd = *rexpect;
}
// Memory32: For cmpxchg8b, the expected value and the result are both in // edx:eax, and the replacement value is in ecx:ebx. But we can't allocate ebx // initially because we need it later for a scratch, so instead we allocate a // temp to hold the low word of 'new'.
// Memory64: Register allocation is particularly hairy here. With memory64, we // have up to seven live values: i64 expected-value, i64 new-value, i64 pointer, // and instance. The instance can use the scratch but there's no avoiding that // we'll run out of registers. // // Unlike for the rmw ops, we can't use edx as the rnew.low since it's used // for the rexpect.high. And we can't push anything onto the stack while we're // popping the memory address because the memory address may be on the stack.
template <> void PopAndAllocate<RegI64>(BaseCompiler* bc, RegI64* rexpect, RegI64* rnew,
RegI64* rd) { // We reserve these (and ebx). The 64-bit pointer will end up in esi+edi.
bc->needI32(bc->specific_.eax);
bc->needI32(bc->specific_.ecx);
bc->needI32(bc->specific_.edx);
// Pop the 'new' value and stash it in the instance scratch area. Do not // initialize *rnew to anything.
RegI64 tmp(Register64(bc->specific_.ecx, bc->specific_.edx));
bc->popI64ToSpecific(tmp);
{
ScratchPtr instanceScratch(*bc);
bc->stashI64(instanceScratch, tmp);
}
template <> void Deallocate<RegI64>(BaseCompiler* bc, RegI64 rexpect, RegI64 rnew) { // edx:ebx have been pushed as the result, and the pointer was freed // separately in the caller, so just free ecx.
bc->free(bc->specific_.ecx);
}
#elifdefined(JS_CODEGEN_ARM)
template <typename RegAddressType> staticvoid PopAndAllocate(BaseCompiler* bc, RegI64* rexpect, RegI64* rnew,
RegI64* rd) { // The replacement value and the result must both be odd/even pairs.
*rnew = bc->popI64Pair();
*rexpect = bc->popI64();
*rd = bc->needI64Pair();
}
// This function assumes a memory index of zero
uint32_t memoryIndex = 0;
int32_t signedLength;
MOZ_ALWAYS_TRUE(popConst(&signedLength));
uint32_t length = signedLength;
MOZ_ASSERT(length != 0 && length <= MaxInlineMemoryCopyLength);
RegI32 src = popI32();
RegI32 dest = popI32();
// Compute the number of copies of each width we will need to do
size_t remainder = length; #ifdef ENABLE_WASM_SIMD
size_t numCopies16 = 0; if (MacroAssembler::SupportsFastUnalignedFPAccesses()) {
numCopies16 = remainder / sizeof(V128);
remainder %= sizeof(V128);
} #endif #ifdef JS_64BIT
size_t numCopies8 = remainder / sizeof(uint64_t);
remainder %= sizeof(uint64_t); #endif
size_t numCopies4 = remainder / sizeof(uint32_t);
remainder %= sizeof(uint32_t);
size_t numCopies2 = remainder / sizeof(uint16_t);
remainder %= sizeof(uint16_t);
size_t numCopies1 = remainder;
// Load all source bytes onto the value stack from low to high using the // widest transfer width we can for the system. We will trap without writing // anything if any source byte is out-of-bounds. bool omitBoundsCheck = false;
size_t offset = 0;
#ifdef ENABLE_WASM_SIMD for (uint32_t i = 0; i < numCopies16; i++) {
RegI32 temp = needI32();
moveI32(src, temp);
pushI32(temp);
// Store all source bytes from the value stack to the destination from // high to low. We will trap without writing anything on the first store // if any dest byte is out-of-bounds.
offset = length;
omitBoundsCheck = false;
// This function assumes a memory index of zero
uint32_t memoryIndex = 0;
int32_t signedLength;
int32_t signedValue;
MOZ_ALWAYS_TRUE(popConst(&signedLength));
MOZ_ALWAYS_TRUE(popConst(&signedValue));
uint32_t length = uint32_t(signedLength);
uint32_t value = uint32_t(signedValue);
MOZ_ASSERT(length != 0 && length <= MaxInlineMemoryFillLength);
RegI32 dest = popI32();
// Compute the number of copies of each width we will need to do
size_t remainder = length; #ifdef ENABLE_WASM_SIMD
size_t numCopies16 = 0; if (MacroAssembler::SupportsFastUnalignedFPAccesses()) {
numCopies16 = remainder / sizeof(V128);
remainder %= sizeof(V128);
} #endif #ifdef JS_64BIT
size_t numCopies8 = remainder / sizeof(uint64_t);
remainder %= sizeof(uint64_t); #endif
size_t numCopies4 = remainder / sizeof(uint32_t);
remainder %= sizeof(uint32_t);
size_t numCopies2 = remainder / sizeof(uint16_t);
remainder %= sizeof(uint16_t);
size_t numCopies1 = remainder;
// Store the fill value to the destination from high to low. We will trap // without writing anything on the first store if any dest byte is // out-of-bounds.
size_t offset = length; bool omitBoundsCheck = false;
////////////////////////////////////////////////////////////////////////////// // // SIMD and Relaxed SIMD.
#ifdef ENABLE_WASM_SIMD void BaseCompiler::loadSplat(MemoryAccessDesc* access) { // We can implement loadSplat mostly as load + splat because the push of the // result onto the value stack in loadCommon normally will not generate any // code, it will leave the value in a register which we will consume.
// We use uint types when we can on the general assumption that unsigned loads // might be smaller/faster on some platforms, because no sign extension needs // to be done after the sub-register load.
RegV128 rd = needV128(); switch (access->type()) { case Scalar::Uint8: {
loadCommon(access, AccessCheck(), ValType::I32);
RegI32 rs = popI32();
masm.splatX16(rs, rd);
free(rs); break;
} case Scalar::Uint16: {
loadCommon(access, AccessCheck(), ValType::I32);
RegI32 rs = popI32();
masm.splatX8(rs, rd);
free(rs); break;
} case Scalar::Uint32: {
loadCommon(access, AccessCheck(), ValType::I32);
RegI32 rs = popI32();
masm.splatX4(rs, rd);
free(rs); break;
} case Scalar::Int64: {
loadCommon(access, AccessCheck(), ValType::I64);
RegI64 rs = popI64();
masm.splatX2(rs, rd);
free(rs); break;
} default:
MOZ_CRASH();
}
pushV128(rd);
}
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.63Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-09-30)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.