Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 8 additions & 7 deletions src/coreclr/jit/codegenxarch.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1041,10 +1041,8 @@ void CodeGen::genCodeForBinary(GenTreeOp* treeNode)
// we can convert it into reg1 = reg1 op reg2 and emit
// the same code as above. Or we need both operands to
// be the same local.
else if (op2reg == targetReg)
else if ((op2reg == targetReg) && (GenTree::OperIsCommutative(oper) || genIsSameLocalVar(op1, op2)))
{
assert(GenTree::OperIsCommutative(oper) || genIsSameLocalVar(op1, op2));

dst = op2;
src = op1;
}
Expand Down Expand Up @@ -1073,10 +1071,14 @@ void CodeGen::genCodeForBinary(GenTreeOp* treeNode)
else
{
// when reg3 != reg1 && reg3 != reg2, and NDD is available, we can use APX-EVEX.ND to optimize the codegen.
eligibleForNDD = emit->DoJitUseApxNDD(ins);
// LSRA also allows reg3 == reg2 for a non-commutative op, but only when it is emitted as NDD.
eligibleForNDD = emit->DoJitUseApxNDD(ins, op2);
if (!eligibleForNDD)
{
var_types op1Type = op1->TypeGet();
// The mov must not clobber op2: its register, or a base/index register of a contained op2.
noway_assert(op2reg != targetReg);
assert((op2->gtGetContainedRegMask() & genRegMask(targetReg)) == 0);
inst_Mov(op1Type, targetReg, op1reg, /* canSkip */ false);
regSet.verifyRegUsed(targetReg);
gcInfo.gcMarkRegPtrVal(targetReg, op1Type);
Expand Down Expand Up @@ -1116,8 +1118,7 @@ void CodeGen::genCodeForBinary(GenTreeOp* treeNode)
// operands should be already formatted above
assert(dst->isUsedFromReg());
assert(op1reg != targetReg);
assert(op2reg != targetReg);
r = emit->emitIns_BASE_R_R_RM(ins, emitTypeSize(treeNode), targetReg, treeNode, dst, src);
r = emit->emitIns_BASE_R_R_RM(ins, emitTypeSize(treeNode), targetReg, treeNode, dst, src, eligibleForNDD);
}
else
{
Expand Down Expand Up @@ -1276,7 +1277,7 @@ void CodeGen::genCodeForMul(GenTreeOp* treeNode)
}
assert(regOp->isUsedFromReg());

emit->emitIns_BASE_R_R_RM(ins, size, mulTargetReg, treeNode, regOp, rmOp);
emit->emitIns_BASE_R_R_RM(ins, size, mulTargetReg, treeNode, regOp, rmOp, emit->DoJitUseApxNDD(ins, rmOp));

// Move the result to the desired register, if necessary
if (ins == INS_mulEAX)
Expand Down
62 changes: 51 additions & 11 deletions src/coreclr/jit/emitxarch.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -1028,6 +1028,26 @@ bool emitter::DoJitUseApxNDD(instruction ins) const
#endif
}

//------------------------------------------------------------------------
// DoJitUseApxNDD: Answer the question: does JIT use APX NDD feature on the given instruction,
// given the r/m operand it would encode?
//
// Arguments:
// ins - instruction to test
// rmOp - the operand that would be encoded as the r/m source
//
// Return Value:
// true if JIT allows APX NDD to be applied on the instruction.
//
bool emitter::DoJitUseApxNDD(instruction ins, GenTree* rmOp) const
{
#if !defined(TARGET_AMD64)
return false;
#else
return DoJitUseApxNDD(ins) && !rmOp->isUsedFromMemory();
#endif
}

inline bool emitter::IsApxConditionalInstruction(instruction ins)
{
#ifdef TARGET_AMD64
Expand Down Expand Up @@ -6506,12 +6526,9 @@ regNumber emitter::emitInsBinary(instruction ins, emitAttr attr, GenTree* dst, G
assert(IsApxNddEncodableInstruction(ins));
// targetReg has to be an actual register if using NDD.
assert(targetReg < REG_STK);
// make sure target register is not either of the src registers.
// targetReg may be src's register (a non-commutative sub), but never dst's.
assert(dst->isUsedFromReg());
regNumber dstreg = dst->GetRegNum();
regNumber srcreg = src->isUsedFromReg() ? src->GetRegNum() : REG_NA;
assert(targetReg != dstreg);
assert(targetReg != srcreg);
assert(targetReg != dst->GetRegNum());
}
#endif

Expand Down Expand Up @@ -10464,22 +10481,45 @@ void emitter::emitIns_BASE_R_R_I(instruction ins, emitAttr attr, regNumber op1Re
}
}

regNumber emitter::emitIns_BASE_R_R_RM(
instruction ins, emitAttr attr, regNumber targetReg, GenTree* treeNode, GenTree* regOp, GenTree* rmOp)
//------------------------------------------------------------------------
// emitIns_BASE_R_R_RM: Emit a binary instruction with a register and an r/m source into targetReg.
//
// Arguments:
// ins - the instruction to emit
// attr - the instruction operand size
// targetReg - the destination register
// treeNode - the node being generated
// regOp - the operand that is in a register
// rmOp - the operand that may be contained (register, memory or immediate)
// useApxNdd - true to emit the non-destructive `ins targetReg, regOp, rmOp` form
//
regNumber emitter::emitIns_BASE_R_R_RM(instruction ins,
emitAttr attr,
regNumber targetReg,
GenTree* treeNode,
GenTree* regOp,
GenTree* rmOp,
bool useApxNdd)
{
bool requiresOverflowCheck = treeNode->gtOverflowEx();
regNumber r = REG_NA;
assert(regOp->isUsedFromReg());
assert(!useApxNdd || DoJitUseApxNDD(ins, rmOp));

bool useApxNdd = DoJitUseApxNDD(ins);
#ifdef DEBUG
if (!useApxNdd && (targetReg != regOp->GetRegNum()))
{
// The `mov` below must not clobber a base/index register of rmOp. LSRA guarantees this:
// isRMWRegOper reports RMW for these nodes, so BuildRMWUses routes the contained operand
// through BuildDelayFreeUses, which keeps its address registers live past the def.
assert((rmOp->gtGetContainedRegMask() & genRegMask(targetReg)) == 0);
}
#endif // DEBUG

if (emitIns_Mov(INS_mov, attr, targetReg, regOp->GetRegNum(), true, useApxNdd) && useApxNdd)
{
return emitInsBinary(ins, attr, regOp, rmOp, targetReg);
}

return emitInsBinary(ins, attr, treeNode, rmOp);
;
}

//----------------------------------------------------------------------------------------
Expand Down
10 changes: 8 additions & 2 deletions src/coreclr/jit/emitxarch.h
Original file line number Diff line number Diff line change
Expand Up @@ -145,6 +145,7 @@ static bool IsBitTestInstruction(instruction ins);
bool IsLegacyMap1(code_t code) const;
bool IsSimdVexOrEvexEncodableInstruction(instruction ins) const;
bool DoJitUseApxNDD(instruction ins) const;
bool DoJitUseApxNDD(instruction ins, GenTree* rmOp) const;

code_t insEncodeMIreg(const instrDesc* id, regNumber reg, emitAttr size, code_t code);

Expand Down Expand Up @@ -1246,8 +1247,13 @@ void emitIns_BASE_R_R(instruction ins, emitAttr attr, regNumber op1Reg, regNumbe

void emitIns_BASE_R_R_I(instruction ins, emitAttr attr, regNumber op1Reg, regNumber op2Reg, int ival);

regNumber emitIns_BASE_R_R_RM(
instruction ins, emitAttr attr, regNumber targetReg, GenTree* treeNode, GenTree* regOp, GenTree* rmOp);
regNumber emitIns_BASE_R_R_RM(instruction ins,
emitAttr attr,
regNumber targetReg,
GenTree* treeNode,
GenTree* regOp,
GenTree* rmOp,
bool useApxNdd);

#ifdef TARGET_AMD64
// Is the last instruction emitted a call instruction?
Expand Down
11 changes: 11 additions & 0 deletions src/coreclr/jit/lsraxarch.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -888,6 +888,17 @@ int LinearScan::BuildRMWUses(
{
delayUseOperand = nullptr;
}

#ifdef TARGET_AMD64
if ((delayUseOperand != nullptr) && node->OperIs(GT_SUB) && !varTypeIsFloating(node) && !op2->isContained() &&
m_compiler->GetEmitter()->DoJitUseApxNDD(INS_sub) && op1->OperIs(GT_LCL_VAR) && isCandidateLocalRef(op1) &&
!op1->AsLclVar()->IsLastUse(0))
{
// NDD reads op2 before writing dst. Not when op1 dies: dst should reuse op1's register (legacy sub).
delayUseOperand = nullptr;
}
#endif // TARGET_AMD64

if (delayUseOperand != nullptr)
{
assert(!prefOp1 || delayUseOperand != op1);
Expand Down
Loading