mirror of
https://repo.dactyloidae.xyz/Dactyloidae/UXP.git
synced 2026-09-30 04:17:31 +09:00
Add SSE2 optimisations
This commit is contained in:
parent
6b25e2373b
commit
57eac6fab8
3 changed files with 35 additions and 33 deletions
|
|
@ -95,12 +95,22 @@ OptimizationInfo::compilerWarmUpThreshold(JSScript* script, jsbytecode* pc) cons
|
|||
// threshold to improve the compilation's type information and hopefully
|
||||
// avoid later recompilation.
|
||||
|
||||
if (script->length() > MAX_MAIN_THREAD_SCRIPT_SIZE)
|
||||
warmUpThreshold *= (script->length() / (double) MAX_MAIN_THREAD_SCRIPT_SIZE);
|
||||
if (script->length() > MAX_MAIN_THREAD_SCRIPT_SIZE) {
|
||||
// Avoid pathological thresholds on very large scripts: large warm-up
|
||||
// counts delay optimization too much for hot UI/update code.
|
||||
double ratio = script->length() / (double) MAX_MAIN_THREAD_SCRIPT_SIZE;
|
||||
if (ratio > 4.0)
|
||||
ratio = 4.0;
|
||||
warmUpThreshold *= ratio;
|
||||
}
|
||||
|
||||
uint32_t numLocalsAndArgs = NumLocalsAndArgs(script);
|
||||
if (numLocalsAndArgs > MAX_MAIN_THREAD_LOCALS_AND_ARGS)
|
||||
warmUpThreshold *= (numLocalsAndArgs / (double) MAX_MAIN_THREAD_LOCALS_AND_ARGS);
|
||||
if (numLocalsAndArgs > MAX_MAIN_THREAD_LOCALS_AND_ARGS) {
|
||||
double ratio = numLocalsAndArgs / (double) MAX_MAIN_THREAD_LOCALS_AND_ARGS;
|
||||
if (ratio > 4.0)
|
||||
ratio = 4.0;
|
||||
warmUpThreshold *= ratio;
|
||||
}
|
||||
|
||||
if (!pc || JitOptions.eagerCompilation)
|
||||
return warmUpThreshold;
|
||||
|
|
@ -110,7 +120,21 @@ OptimizationInfo::compilerWarmUpThreshold(JSScript* script, jsbytecode* pc) cons
|
|||
// Note that the loop depth is always > 0 so we will prefer non-OSR over OSR.
|
||||
uint32_t loopDepth = LoopEntryDepthHint(pc);
|
||||
MOZ_ASSERT(loopDepth > 0);
|
||||
return warmUpThreshold + loopDepth * 100;
|
||||
|
||||
// jQuery-style code often executes many small hot loops. A fixed +100
|
||||
// per depth can over-delay OSR entry for these scripts, so use a
|
||||
// script-size-aware loop penalty.
|
||||
uint32_t perDepthPenalty;
|
||||
if (JitOptions.isSmallFunction(script)) {
|
||||
perDepthPenalty = 25;
|
||||
} else {
|
||||
perDepthPenalty = warmUpThreshold / 8;
|
||||
if (perDepthPenalty < 50)
|
||||
perDepthPenalty = 50;
|
||||
if (perDepthPenalty > 200)
|
||||
perDepthPenalty = 200;
|
||||
}
|
||||
return warmUpThreshold + loopDepth * perDepthPenalty;
|
||||
}
|
||||
|
||||
OptimizationLevelInfo::OptimizationLevelInfo()
|
||||
|
|
|
|||
|
|
@ -631,8 +631,7 @@ LIRGeneratorX86::visitInt64ToFloatingPoint(MInt64ToFloatingPoint* ins)
|
|||
MOZ_ASSERT(opd->type() == MIRType::Int64);
|
||||
MOZ_ASSERT(IsFloatingPointType(ins->type()));
|
||||
|
||||
LDefinition maybeTemp =
|
||||
(ins->isUnsigned() && AssemblerX86Shared::HasSSE3()) ? temp() : LDefinition::BogusTemp();
|
||||
LDefinition maybeTemp = LDefinition::BogusTemp();
|
||||
|
||||
define(new(alloc()) LInt64ToFloatingPoint(useInt64Register(opd), maybeTemp), ins);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -25,36 +25,14 @@ static const double TO_DOUBLE_HIGH_SCALE = 0x100000000;
|
|||
bool
|
||||
MacroAssemblerX86::convertUInt64ToDoubleNeedsTemp()
|
||||
{
|
||||
return HasSSE3();
|
||||
return false;
|
||||
}
|
||||
|
||||
void
|
||||
MacroAssemblerX86::convertUInt64ToDouble(Register64 src, FloatRegister dest, Register temp)
|
||||
{
|
||||
// SUBPD needs SSE2, HADDPD needs SSE3.
|
||||
if (!HasSSE3()) {
|
||||
MOZ_ASSERT(temp == Register::Invalid());
|
||||
|
||||
// Zero the dest register to break dependencies, see convertInt32ToDouble.
|
||||
zeroDouble(dest);
|
||||
|
||||
asMasm().Push(src.high);
|
||||
asMasm().Push(src.low);
|
||||
fild(Operand(esp, 0));
|
||||
|
||||
Label notNegative;
|
||||
asMasm().branch32(Assembler::NotSigned, src.high, Imm32(0), ¬Negative);
|
||||
double add_constant = 18446744073709551616.0; // 2^64
|
||||
store64(Imm64(mozilla::BitwiseCast<uint64_t>(add_constant)), Address(esp, 0));
|
||||
fld(Operand(esp, 0));
|
||||
faddp();
|
||||
bind(¬Negative);
|
||||
|
||||
fstp(Operand(esp, 0));
|
||||
vmovsd(Address(esp, 0), dest);
|
||||
asMasm().freeStack(2*sizeof(intptr_t));
|
||||
return;
|
||||
}
|
||||
(void) temp;
|
||||
MOZ_ASSERT(HasSSE2());
|
||||
|
||||
// Following operation uses entire 128-bit of dest XMM register.
|
||||
// Currently higher 64-bit is free when we have access to lower 64-bit.
|
||||
|
|
@ -113,7 +91,8 @@ MacroAssemblerX86::convertUInt64ToDouble(Register64 src, FloatRegister dest, Reg
|
|||
// LO(dest) = double(0x HHHHHHHH 00000000) + double(0x 00000000 LLLLLLLL)
|
||||
// = double(0x HHHHHHHH LLLLLLLL)
|
||||
// = double(src)
|
||||
vhaddpd(dest128, dest128);
|
||||
vmovhlps(dest128, dest128, ScratchSimd128Reg);
|
||||
vaddsd(ScratchSimd128Reg, dest128, dest128);
|
||||
}
|
||||
|
||||
void
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue