This is an archive of the discontinued LLVM Phabricator instance.

AMDGPU: Allow vectorization of round intrinsic
ClosedPublic

Authored by arsenm on Mar 20 2020, 2:45 PM.

Download Raw Diff

Details

Reviewers

rampitec
kzhuravl
vpykhtin
dfukalov
aartbik

Summary

There seems to be a small benefit to the legalized sequence for v2f16
round with packed instructions, so allow vectorizing it by reducing
the cost.

An unintended side effect is vectorization of f32 round also
happens. The current FMA logic seems off to me, and isn't checking for
packed instructions.

Diff Detail

Event Timeline

arsenm created this revision.Mar 20 2020, 2:45 PM

Herald added a reviewer: aartbik. · View Herald TranscriptMar 20 2020, 2:45 PM

Herald added subscribers: kerbowa, hiraditya, t-tye and 6 others. · View Herald Transcript

rampitec accepted this revision.Mar 23 2020, 1:35 PM

This revision is now accepted and ready to land.Mar 23 2020, 1:35 PM

66073953a5cb7ecac32bcb897033fdc337c56b5e

Revision Contents

Path

Size

llvm/

lib/

Target/

AMDGPU/

AMDGPUTargetTransformInfo.cpp

25 lines

test/

Transforms/

SLPVectorizer/

AMDGPU/

round.ll

38 lines

Diff 251775

llvm/lib/Target/AMDGPU/AMDGPUTargetTransformInfo.cpp

Show First 20 Lines • Show All 470 Lines • ▼ Show 20 Lines	int GCNTTIImpl::getArithmeticInstrCost(unsigned Opcode, Type *Ty,
default:		default:
break;		break;
}		}

return BaseT::getArithmeticInstrCost(Opcode, Ty, Opd1Info, Opd2Info,		return BaseT::getArithmeticInstrCost(Opcode, Ty, Opd1Info, Opd2Info,
Opd1PropInfo, Opd2PropInfo);		Opd1PropInfo, Opd2PropInfo);
}		}

		// Return true if there's a potential benefit from using v2f16 instructions for
		// an intrinsic, even if it requires nontrivial legalization.
		static bool intrinsicHasPackedVectorBenefit(Intrinsic::ID ID) {
		switch (ID) {
		case Intrinsic::fma: // TODO: fmuladd
		// There's a small benefit to using vector ops in the legalized code.
		case Intrinsic::round:
		return true;
		default:
		return false;
		}
		}

template <typename T>		template <typename T>
int GCNTTIImpl::getIntrinsicInstrCost(Intrinsic::ID ID, Type *RetTy,		int GCNTTIImpl::getIntrinsicInstrCost(Intrinsic::ID ID, Type *RetTy,
ArrayRef<T *> Args, FastMathFlags FMF,		ArrayRef<T *> Args, FastMathFlags FMF,
unsigned VF, const Instruction *I) {		unsigned VF, const Instruction *I) {
if (ID != Intrinsic::fma)		if (!intrinsicHasPackedVectorBenefit(ID))
return BaseT::getIntrinsicInstrCost(ID, RetTy, Args, FMF, VF, I);		return BaseT::getIntrinsicInstrCost(ID, RetTy, Args, FMF, VF, I);

EVT OrigTy = TLI->getValueType(DL, RetTy);		EVT OrigTy = TLI->getValueType(DL, RetTy);
if (!OrigTy.isSimple()) {		if (!OrigTy.isSimple()) {
return BaseT::getIntrinsicInstrCost(ID, RetTy, Args, FMF, VF, I);		return BaseT::getIntrinsicInstrCost(ID, RetTy, Args, FMF, VF, I);
}		}

// Legalize the type.		// Legalize the type.
std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, RetTy);		std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, RetTy);

unsigned NElts = LT.second.isVector() ?		unsigned NElts = LT.second.isVector() ?
LT.second.getVectorNumElements() : 1;		LT.second.getVectorNumElements() : 1;

MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;		MVT::SimpleValueType SLT = LT.second.getScalarType().SimpleTy;

if (SLT == MVT::f64)		if (SLT == MVT::f64)
return LT.first * NElts * get64BitInstrCost();		return LT.first * NElts * get64BitInstrCost();

if (ST->has16BitInsts() && SLT == MVT::f16)		if (ST->has16BitInsts() && SLT == MVT::f16)
NElts = (NElts + 1) / 2;		NElts = (NElts + 1) / 2;

return LT.first * NElts * (ST->hasFastFMAF32() ? getHalfRateInstrCost()		// TODO: Get more refined intrinsic costs?
: getQuarterRateInstrCost());		unsigned InstRate = getQuarterRateInstrCost();
		if (ID == Intrinsic::fma) {
		InstRate = ST->hasFastFMAF32() ? getHalfRateInstrCost()
		: getQuarterRateInstrCost();
		}

		return LT.first * NElts * InstRate;
}		}

int GCNTTIImpl::getIntrinsicInstrCost(Intrinsic::ID ID, Type *RetTy,		int GCNTTIImpl::getIntrinsicInstrCost(Intrinsic::ID ID, Type *RetTy,
ArrayRef<Value *> Args, FastMathFlags FMF,		ArrayRef<Value *> Args, FastMathFlags FMF,
unsigned VF, const Instruction *I) {		unsigned VF, const Instruction *I) {
return getIntrinsicInstrCost<Value>(ID, RetTy, Args, FMF, VF, I);		return getIntrinsicInstrCost<Value>(ID, RetTy, Args, FMF, VF, I);
}		}

▲ Show 20 Lines • Show All 564 Lines • Show Last 20 Lines

llvm/test/Transforms/SLPVectorizer/AMDGPU/round.ll

This file was added.

				; RUN: opt -S -mtriple=amdgcn-amd-amdhsa -mcpu=hawaii -slp-vectorizer %s \| FileCheck -check-prefixes=GCN,GFX7 %s
				; RUN: opt -S -mtriple=amdgcn-amd-amdhsa -mcpu=fiji -slp-vectorizer %s \| FileCheck -check-prefixes=GCN,GFX8 %s
				; RUN: opt -S -mtriple=amdgcn-amd-amdhsa -mcpu=gfx900 -slp-vectorizer %s \| FileCheck -check-prefixes=GCN,GFX8 %s

				; GCN-LABEL: @round_v2f16(
				; GFX7: call half @llvm.round.f16(
				; GFX7: call half @llvm.round.f16(

				; GFX8: call <2 x half> @llvm.round.v2f16(
				define <2 x half> @round_v2f16(<2 x half> %arg) {
				bb:
				%tmp = extractelement <2 x half> %arg, i64 0
				%tmp1 = tail call half @llvm.round.half(half %tmp)
				%tmp2 = insertelement <2 x half> undef, half %tmp1, i64 0
				%tmp3 = extractelement <2 x half> %arg, i64 1
				%tmp4 = tail call half @llvm.round.half(half %tmp3)
				%tmp5 = insertelement <2 x half> %tmp2, half %tmp4, i64 1
				ret <2 x half> %tmp5
				}

				; TODO: Should probably not really be vectorizing this
				; GCN-LABEL: @round_v2f32(
				; GCN: call <2 x float> @llvm.round.v2f32
				define <2 x float> @round_v2f32(<2 x float> %arg) {
				bb:
				%tmp = extractelement <2 x float> %arg, i64 0
				%tmp1 = tail call float @llvm.round.f32(float %tmp)
				%tmp2 = insertelement <2 x float> undef, float %tmp1, i64 0
				%tmp3 = extractelement <2 x float> %arg, i64 1
				%tmp4 = tail call float @llvm.round.f32(float %tmp3)
				%tmp5 = insertelement <2 x float> %tmp2, float %tmp4, i64 1
				ret <2 x float> %tmp5
				}

				declare half @llvm.round.half(half) #0
				declare float @llvm.round.f32(float) #0

				attributes #0 = { nounwind readnone speculatable willreturn }