Add SSE2 optimisations

This commit is contained in:
ownedbywuigi 2026-03-28 09:18:09 +00:00
commit 57eac6fab8
3 changed files with 35 additions and 33 deletions

View file

@ -95,12 +95,22 @@ OptimizationInfo::compilerWarmUpThreshold(JSScript* script, jsbytecode* pc) cons
// threshold to improve the compilation's type information and hopefully
// avoid later recompilation.
if (script->length() > MAX_MAIN_THREAD_SCRIPT_SIZE)
warmUpThreshold *= (script->length() / (double) MAX_MAIN_THREAD_SCRIPT_SIZE);
if (script->length() > MAX_MAIN_THREAD_SCRIPT_SIZE) {
// Avoid pathological thresholds on very large scripts: large warm-up
// counts delay optimization too much for hot UI/update code.
double ratio = script->length() / (double) MAX_MAIN_THREAD_SCRIPT_SIZE;
if (ratio > 4.0)
ratio = 4.0;
warmUpThreshold *= ratio;
}
uint32_t numLocalsAndArgs = NumLocalsAndArgs(script);
if (numLocalsAndArgs > MAX_MAIN_THREAD_LOCALS_AND_ARGS)
warmUpThreshold *= (numLocalsAndArgs / (double) MAX_MAIN_THREAD_LOCALS_AND_ARGS);
if (numLocalsAndArgs > MAX_MAIN_THREAD_LOCALS_AND_ARGS) {
double ratio = numLocalsAndArgs / (double) MAX_MAIN_THREAD_LOCALS_AND_ARGS;
if (ratio > 4.0)
ratio = 4.0;
warmUpThreshold *= ratio;
}
if (!pc || JitOptions.eagerCompilation)
return warmUpThreshold;
@ -110,7 +120,21 @@ OptimizationInfo::compilerWarmUpThreshold(JSScript* script, jsbytecode* pc) cons
// Note that the loop depth is always > 0 so we will prefer non-OSR over OSR.
uint32_t loopDepth = LoopEntryDepthHint(pc);
MOZ_ASSERT(loopDepth > 0);
return warmUpThreshold + loopDepth * 100;
// jQuery-style code often executes many small hot loops. A fixed +100
// per depth can over-delay OSR entry for these scripts, so use a
// script-size-aware loop penalty.
uint32_t perDepthPenalty;
if (JitOptions.isSmallFunction(script)) {
perDepthPenalty = 25;
} else {
perDepthPenalty = warmUpThreshold / 8;
if (perDepthPenalty < 50)
perDepthPenalty = 50;
if (perDepthPenalty > 200)
perDepthPenalty = 200;
}
return warmUpThreshold + loopDepth * perDepthPenalty;
}
OptimizationLevelInfo::OptimizationLevelInfo()

View file

@ -631,8 +631,7 @@ LIRGeneratorX86::visitInt64ToFloatingPoint(MInt64ToFloatingPoint* ins)
MOZ_ASSERT(opd->type() == MIRType::Int64);
MOZ_ASSERT(IsFloatingPointType(ins->type()));
LDefinition maybeTemp =
(ins->isUnsigned() && AssemblerX86Shared::HasSSE3()) ? temp() : LDefinition::BogusTemp();
LDefinition maybeTemp = LDefinition::BogusTemp();
define(new(alloc()) LInt64ToFloatingPoint(useInt64Register(opd), maybeTemp), ins);
}

View file

@ -25,36 +25,14 @@ static const double TO_DOUBLE_HIGH_SCALE = 0x100000000;
bool
MacroAssemblerX86::convertUInt64ToDoubleNeedsTemp()
{
return HasSSE3();
return false;
}
void
MacroAssemblerX86::convertUInt64ToDouble(Register64 src, FloatRegister dest, Register temp)
{
// SUBPD needs SSE2, HADDPD needs SSE3.
if (!HasSSE3()) {
MOZ_ASSERT(temp == Register::Invalid());
// Zero the dest register to break dependencies, see convertInt32ToDouble.
zeroDouble(dest);
asMasm().Push(src.high);
asMasm().Push(src.low);
fild(Operand(esp, 0));
Label notNegative;
asMasm().branch32(Assembler::NotSigned, src.high, Imm32(0), &notNegative);
double add_constant = 18446744073709551616.0; // 2^64
store64(Imm64(mozilla::BitwiseCast<uint64_t>(add_constant)), Address(esp, 0));
fld(Operand(esp, 0));
faddp();
bind(&notNegative);
fstp(Operand(esp, 0));
vmovsd(Address(esp, 0), dest);
asMasm().freeStack(2*sizeof(intptr_t));
return;
}
(void) temp;
MOZ_ASSERT(HasSSE2());
// Following operation uses entire 128-bit of dest XMM register.
// Currently higher 64-bit is free when we have access to lower 64-bit.
@ -113,7 +91,8 @@ MacroAssemblerX86::convertUInt64ToDouble(Register64 src, FloatRegister dest, Reg
// LO(dest) = double(0x HHHHHHHH 00000000) + double(0x 00000000 LLLLLLLL)
// = double(0x HHHHHHHH LLLLLLLL)
// = double(src)
vhaddpd(dest128, dest128);
vmovhlps(dest128, dest128, ScratchSimd128Reg);
vaddsd(ScratchSimd128Reg, dest128, dest128);
}
void