LLVM 24.0.0git
ARMTargetTransformInfo.h
Go to the documentation of this file.
1//===- ARMTargetTransformInfo.h - ARM specific TTI --------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file a TargetTransformInfoImplBase conforming object specific to the
11/// ARM target machine. It uses the target's detailed information to
12/// provide more precise answers to certain TTI queries, while letting the
13/// target independent and default TTI implementations handle the rest.
14//
15//===----------------------------------------------------------------------===//
16
17#ifndef LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
18#define LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
19
20#include "ARM.h"
21#include "ARMSubtarget.h"
22#include "ARMTargetMachine.h"
23#include "llvm/ADT/ArrayRef.h"
26#include "llvm/IR/Constant.h"
27#include "llvm/IR/Function.h"
29#include <optional>
30
31namespace llvm {
32
33class APInt;
35class Instruction;
36class Loop;
37class SCEV;
38class ScalarEvolution;
39class Type;
40class Value;
41
51
52// For controlling conversion of memcpy into Tail Predicated loop.
53namespace TPLoop {
55}
56
57class ARMTTIImpl final : public BasicTTIImplBase<ARMTTIImpl> {
58 using BaseT = BasicTTIImplBase<ARMTTIImpl>;
59 using TTI = TargetTransformInfo;
60
61 friend BaseT;
62
63 const ARMSubtarget *ST;
64 const ARMTargetLowering *TLI;
65
66 const ARMSubtarget *getST() const { return ST; }
67 const ARMTargetLowering *getTLI() const { return TLI; }
68
69public:
70 explicit ARMTTIImpl(const ARMBaseTargetMachine *TM, const Function &F)
71 : BaseT(TM, F.getDataLayout()), ST(TM->getSubtargetImpl(F)),
72 TLI(ST->getTargetLowering()) {}
73
74 bool enableInterleavedAccessVectorization() const override { return true; }
75
77 getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override;
78
79 /// Floating-point computation using ARMv8 AArch32 Advanced
80 /// SIMD instructions remains unchanged from ARMv7. Only AArch64 SIMD
81 /// and Arm MVE are IEEE-754 compliant.
82 bool isFPVectorizationPotentiallyUnsafe() const override {
83 return !ST->isTargetDarwin() && !ST->hasMVEFloatOps();
84 }
85
86 std::optional<Instruction *>
88 std::optional<Value *> simplifyDemandedVectorEltsIntrinsic(
89 InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts,
90 APInt &UndefElts2, APInt &UndefElts3,
91 std::function<void(Instruction *, unsigned, APInt, APInt &)>
92 SimplifyAndSetOp) const override;
93
94 /// \name Scalar TTI Implementations
95 /// @{
96
97 InstructionCost getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx,
98 const APInt &Imm,
99 Type *Ty) const override;
100
103 TTI::TargetCostKind CostKind) const override;
104
105 InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx,
106 const APInt &Imm, Type *Ty,
108 Instruction *Inst = nullptr) const override;
109
110 /// @}
111
112 /// \name Vector TTI Implementations
113 /// @{
114
115 unsigned getNumberOfRegisters(unsigned ClassID) const override {
116 bool Vector = (ClassID == 1);
117 if (Vector) {
118 if (ST->hasNEON())
119 return 16;
120 if (ST->hasMVEIntegerOps())
121 return 8;
122 return 0;
123 }
124
125 if (ST->isThumb1Only())
126 return 8;
127 return 13;
128 }
129
132 switch (K) {
134 return TypeSize::getFixed(32);
136 if (ST->hasNEON())
137 return TypeSize::getFixed(128);
138 if (ST->hasMVEIntegerOps())
139 return TypeSize::getFixed(128);
140 return TypeSize::getFixed(0);
142 return TypeSize::getScalable(0);
143 }
144 llvm_unreachable("Unsupported register kind");
145 }
146
148 bool HasUnorderedReductions) const override {
149 return ST->getMaxInterleaveFactor();
150 }
151
152 bool isProfitableLSRChainElement(Instruction *I) const override;
153
154 bool
155 isLegalMaskedLoad(Type *DataTy, Align Alignment, unsigned AddressSpace,
156 TTI::MaskKind MaskKind =
158
159 bool
160 isLegalMaskedStore(Type *DataTy, Align Alignment, unsigned AddressSpace,
161 TTI::MaskKind MaskKind =
163 return isLegalMaskedLoad(DataTy, Alignment, AddressSpace, MaskKind);
164 }
165
167 Align Alignment) const override {
168 // For MVE, we have a custom lowering pass that will already have custom
169 // legalised any gathers that we can lower to MVE intrinsics, and want to
170 // expand all the rest. The pass runs before the masked intrinsic lowering
171 // pass.
172 return true;
173 }
174
176 Align Alignment) const override {
177 return forceScalarizeMaskedGather(VTy, Alignment);
178 }
179
180 bool isLegalMaskedGather(Type *Ty, Align Alignment) const override;
181
182 bool isLegalMaskedScatter(Type *Ty, Align Alignment) const override {
183 return isLegalMaskedGather(Ty, Alignment);
184 }
185
186 InstructionCost getMemcpyCost(const Instruction *I) const override;
187
189 return ST->getMaxInlineSizeThreshold();
190 }
191
192 int getNumMemOps(const IntrinsicInst *I) const;
193
197 VectorType *SubTp, ArrayRef<const Value *> Args = {},
198 const Instruction *CtxI = nullptr,
200 TTI::VectorInstrContext::None) const override;
201
202 bool preferInLoopReduction(RecurKind Kind, Type *Ty) const override;
203
204 bool preferPredicatedReductionSelect() const override;
205
206 bool shouldExpandReduction(const IntrinsicInst *II) const override {
207 return false;
208 }
209
211 const Instruction *I = nullptr) const override;
212
214 getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src,
216 const Instruction *I = nullptr) const override;
217
219 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
223 const Instruction *I = nullptr) const override;
224
228 unsigned Index, const Value *Op0, const Value *Op1,
230 TTI::VectorInstrContext::None) const override;
231
233 getAddressComputationCost(Type *Ty, ScalarEvolution *SE, const SCEV *Ptr,
234 TTI::TargetCostKind CostKind) const override;
235
237 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
241 const Instruction *CtxI = nullptr) const override;
242
244 unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace,
247 const Instruction *I = nullptr) const override;
248
250 getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA,
251 TTI::TargetCostKind CostKind) const override;
252
253 InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA,
255
257 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
258 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
259 bool UseMaskForCond = false, bool UseMaskForGaps = false) const override;
260
261 InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA,
263
265 getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy,
266 std::optional<FastMathFlags> FMF,
267 TTI::TargetCostKind CostKind) const override;
269 getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy,
270 VectorType *ValTy, std::optional<FastMathFlags> FMF,
271 TTI::TargetCostKind CostKind) const override;
273 getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy,
274 VectorType *ValTy,
275 TTI::TargetCostKind CostKind) const override;
276
278 getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF,
279 TTI::TargetCostKind CostKind) const override;
280
282 getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
283 TTI::TargetCostKind CostKind) const override;
284
286 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
288 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
290 std::optional<FastMathFlags> FMF) const override {
292 }
293
294 /// getScalingFactorCost - Return the cost of the scaling used in
295 /// addressing mode represented by AM.
296 /// If the AM is supported, the return value must be >= 0.
297 /// If the AM is not supported, the return value is an invalid cost.
299 StackOffset BaseOffset, bool HasBaseReg,
300 int64_t Scale,
301 unsigned AddrSpace) const override;
302
303 bool maybeLoweredToCall(Instruction &I) const;
304 bool isLoweredToCall(const Function *F) const override;
307 HardwareLoopInfo &HWLoopInfo) const override;
308 bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override;
311 OptimizationRemarkEmitter *ORE) const override;
312
314
316 TTI::PeelingPreferences &PP) const override;
318 // In the ROPI and RWPI relocation models we can't have pointers to global
319 // variables or functions in constant data, so don't convert switches to
320 // lookup tables if any of the values would need relocation.
321 if (ST->isROPI() || ST->isRWPI())
322 return !C->needsDynamicRelocation();
323
324 return true;
325 }
326
327 bool shouldConsiderVectorizationRegPressure() const override;
328
329 bool hasArmWideBranch(bool Thumb) const override;
330
332 SmallVectorImpl<Use *> &Ops) const override;
333
334 unsigned getNumBytesToPadGlobalArray(unsigned Size,
335 Type *ArrayType) const override;
336
337 /// @}
338};
339
340/// isVREVMask - Check if a vector shuffle corresponds to a VREV
341/// instruction with the specified blocksize. (The order of the elements
342/// within each block of the vector is reversed.)
343inline bool isVREVMask(ArrayRef<int> M, EVT VT, unsigned BlockSize) {
344 assert((BlockSize == 16 || BlockSize == 32 || BlockSize == 64) &&
345 "Only possible block sizes for VREV are: 16, 32, 64");
346
347 unsigned EltSz = VT.getScalarSizeInBits();
348 if (EltSz != 8 && EltSz != 16 && EltSz != 32)
349 return false;
350
351 unsigned BlockElts = M[0] + 1;
352 // If the first shuffle index is UNDEF, be optimistic.
353 if (M[0] < 0)
354 BlockElts = BlockSize / EltSz;
355
356 if (BlockSize <= EltSz || BlockSize != BlockElts * EltSz)
357 return false;
358
359 for (unsigned i = 0, e = M.size(); i < e; ++i) {
360 if (M[i] < 0)
361 continue; // ignore UNDEF indices
362 if ((unsigned)M[i] != (i - i % BlockElts) + (BlockElts - 1 - i % BlockElts))
363 return false;
364 }
365
366 return true;
367}
368
369} // end namespace llvm
370
371#endif // LLVM_LIB_TARGET_ARM_ARMTARGETTRANSFORMINFO_H
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
uint64_t IntrinsicInst * II
static const int BlockSize
Definition TarWriter.cpp:33
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isFPVectorizationPotentiallyUnsafe() const override
Floating-point computation using ARMv8 AArch32 Advanced SIMD instructions remains unchanged from ARMv...
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getMemcpyCost(const Instruction *I) const override
bool isLegalMaskedScatter(Type *Ty, Align Alignment) const override
bool maybeLoweredToCall(Instruction &I) const
bool preferInLoopReduction(RecurKind Kind, Type *Ty) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *ValTy, TTI::TargetCostKind CostKind) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool hasArmWideBranch(bool Thumb) const override
bool shouldConsiderVectorizationRegPressure() const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
bool shouldExpandReduction(const IntrinsicInst *II) const override
bool shouldBuildLookupTablesForConstant(Constant *C) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
int getNumMemOps(const IntrinsicInst *I) const
Given a memcpy/memset/memmove instruction, return the number of memory operations performed,...
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty) const override
bool isLoweredToCall(const Function *F) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
uint64_t getMaxMemIntrinsicInlineSizeThreshold() const override
bool isLegalMaskedStore(Type *DataTy, Align Alignment, unsigned AddressSpace, TTI::MaskKind MaskKind=TTI::MaskKind::VariableOrConstantMask) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
bool forceScalarizeMaskedScatter(VectorType *VTy, Align Alignment) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
bool isLegalMaskedLoad(Type *DataTy, Align Alignment, unsigned AddressSpace, TTI::MaskKind MaskKind=TTI::MaskKind::VariableOrConstantMask) const override
ARMTTIImpl(const ARMBaseTargetMachine *TM, const Function &F)
TailFoldingStyle getPreferredTailFoldingStyle() const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
bool forceScalarizeMaskedGather(VectorType *VTy, Align Alignment) const override
unsigned getNumberOfRegisters(unsigned ClassID) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
bool preferPredicatedReductionSelect() const override
bool isLegalMaskedGather(Type *Ty, Align Alignment) const override
InstructionCost getAddressComputationCost(Type *Ty, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
unsigned getNumBytesToPadGlobalArray(unsigned Size, Type *ArrayType) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isProfitableLSRChainElement(Instruction *I) const override
bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, AssumptionCache &AC, TargetLibraryInfo *LibInfo, HardwareLoopInfo &HWLoopInfo) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool enableInterleavedAccessVectorization() const override
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
getScalingFactorCost - Return the cost of the scaling used in addressing mode represented by AM.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Class to represent array types.
A cache of @llvm.assume calls within a function.
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
BasicTTIImplBase(const TargetMachine *TM, const DataLayout &DL)
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
This is an important base class in LLVM.
Definition Constant.h:43
The core instruction combiner logic.
static InstructionCost getInvalid(CostType Val=0)
A wrapper class for inspecting calls to intrinsic functions.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
The optimization diagnostic interface.
This class represents an analyzed expression in the program.
The main scalar evolution driver.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
Provides information about what library functions are available for the current target.
virtual const DataLayout & getDataLayout() const
virtual InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
MaskKind
Some targets only support masked load/store with a constant mask.
TargetCostKind
The kind of cost model.
llvm::VectorInstrContext VectorInstrContext
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM Value Representation.
Definition Value.h:75
Base class of all SIMD vector types.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
ArrayRef(const T &OneElt) -> ArrayRef< T >
bool isVREVMask(ArrayRef< int > M, EVT VT, unsigned BlockSize)
isVREVMask - Check if a vector shuffle corresponds to a VREV instruction with the specified blocksize...
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
Attributes of a target dependent hardware loop.
Parameters that control the generic loop unrolling transformation.