LLVM 24.0.0git
InstrBuilder.cpp
Go to the documentation of this file.
1//===--------------------- InstrBuilder.cpp ---------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9///
10/// This file implements the InstrBuilder interface.
11///
12//===----------------------------------------------------------------------===//
13
15#include "llvm/ADT/APInt.h"
16#include "llvm/ADT/DenseMap.h"
17#include "llvm/ADT/Hashing.h"
18#include "llvm/ADT/Statistic.h"
19#include "llvm/MC/MCInst.h"
20#include "llvm/MCA/Support.h"
21#include "llvm/Support/Debug.h"
24
25#define DEBUG_TYPE "llvm-mca-instrbuilder"
26
27namespace llvm {
28namespace mca {
29
30char RecycledInstErr::ID = 0;
31
33
34InstrBuilder::InstrBuilder(const llvm::MCSubtargetInfo &sti,
35 const llvm::MCInstrInfo &mcii,
36 const llvm::MCRegisterInfo &mri,
37 const llvm::MCInstrAnalysis *mcia,
38 const mca::InstrumentManager &im, unsigned cl)
39 : STI(sti), MCII(mcii), MRI(mri), MCIA(mcia), IM(im), FirstCallInst(true),
40 FirstReturnInst(true), CallLatency(cl) {
41 const MCSchedModel &SM = STI.getSchedModel();
42 ProcResourceMasks.resize(SM.getNumProcResourceKinds());
43 computeProcResourceMasks(STI.getSchedModel(), ProcResourceMasks);
44}
45
47 const MCSchedClassDesc &SCDesc,
48 const MCSubtargetInfo &STI,
49 ArrayRef<uint64_t> ProcResourceMasks) {
50 const MCSchedModel &SM = STI.getSchedModel();
51
52 // Populate resources consumed.
53 using ResourcePlusCycles = std::pair<uint64_t, ResourceUsage>;
55
56 // Track cycles contributed by resources that are in a "Super" relationship.
57 // This is required if we want to correctly match the behavior of method
58 // SubtargetEmitter::ExpandProcResource() in Tablegen. When computing the set
59 // of "consumed" processor resources and resource cycles, the logic in
60 // ExpandProcResource() doesn't update the number of resource cycles
61 // contributed by a "Super" resource to a group.
62 // We need to take this into account when we find that a processor resource is
63 // part of a group, and it is also used as the "Super" of other resources.
64 // This map stores the number of cycles contributed by sub-resources that are
65 // part of a "Super" resource. The key value is the "Super" resource mask ID.
66 DenseMap<uint64_t, unsigned> SuperResources;
67
68 unsigned NumProcResources = SM.getNumProcResourceKinds();
69 APInt Buffers(NumProcResources, 0);
70
71 bool AllInOrderResources = true;
72 bool AnyDispatchHazards = false;
73 for (unsigned I = 0, E = SCDesc.NumWriteProcResEntries; I < E; ++I) {
74 const MCWriteProcResEntry *PRE = STI.getWriteProcResBegin(&SCDesc) + I;
76 if (!PRE->ReleaseAtCycle) {
77#ifndef NDEBUG
79 << "Ignoring invalid write of zero cycles on processor resource "
80 << PR.Name << "\n";
81 WithColor::note() << "found in scheduling class "
82 << SM.getSchedClassName(ID.SchedClassID)
83 << " (write index #" << I << ")\n";
84#endif
85 continue;
86 }
87
88 uint64_t Mask = ProcResourceMasks[PRE->ProcResourceIdx];
89 const int BufferSize = SM.getResourceBufferSize(PRE->ProcResourceIdx);
90 if (BufferSize < 0) {
91 AllInOrderResources = false;
92 } else {
93 Buffers.setBit(getResourceStateIndex(Mask));
94 AnyDispatchHazards |= (BufferSize == 0);
95 AllInOrderResources &= (BufferSize <= 1);
96 }
97
98 CycleSegment RCy(0, PRE->ReleaseAtCycle, false);
99 Worklist.emplace_back(ResourcePlusCycles(Mask, ResourceUsage(RCy)));
100 if (PR.SuperIdx) {
101 uint64_t Super = ProcResourceMasks[PR.SuperIdx];
102 SuperResources[Super] += PRE->ReleaseAtCycle;
103 }
104 }
105
106 ID.MustIssueImmediately = AllInOrderResources && AnyDispatchHazards;
107
108 // Sort elements by mask popcount, so that we prioritize resource units over
109 // resource groups, and smaller groups over larger groups.
110 sort(Worklist, [](const ResourcePlusCycles &A, const ResourcePlusCycles &B) {
111 unsigned popcntA = llvm::popcount(A.first);
112 unsigned popcntB = llvm::popcount(B.first);
113 if (popcntA < popcntB)
114 return true;
115 if (popcntA > popcntB)
116 return false;
117 return A.first < B.first;
118 });
119
120 uint64_t UsedResourceUnits = 0;
121 uint64_t UsedResourceGroups = 0;
122 uint64_t UnitsFromResourceGroups = 0;
123
124 // Remove cycles contributed by smaller resources, and check if there
125 // are partially overlapping resource groups.
126 ID.HasPartiallyOverlappingGroups = false;
127
128 for (unsigned I = 0, E = Worklist.size(); I < E; ++I) {
129 ResourcePlusCycles &A = Worklist[I];
130 if (!A.second.size()) {
131 assert(llvm::popcount(A.first) > 1 && "Expected a group!");
132 UsedResourceGroups |= llvm::bit_floor(A.first);
133 continue;
134 }
135
136 ID.Resources.emplace_back(A);
137 uint64_t NormalizedMask = A.first;
138
139 if (llvm::popcount(A.first) == 1) {
140 UsedResourceUnits |= A.first;
141 } else {
142 // Remove the leading 1 from the resource group mask.
143 NormalizedMask ^= llvm::bit_floor(NormalizedMask);
144 if (UnitsFromResourceGroups & NormalizedMask)
145 ID.HasPartiallyOverlappingGroups = true;
146
147 UnitsFromResourceGroups |= NormalizedMask;
148 UsedResourceGroups |= (A.first ^ NormalizedMask);
149 }
150
151 for (unsigned J = I + 1; J < E; ++J) {
152 ResourcePlusCycles &B = Worklist[J];
153 if ((NormalizedMask & B.first) == NormalizedMask) {
154 B.second.CS.subtract(A.second.size() - SuperResources[A.first]);
155 if (llvm::popcount(B.first) > 1)
156 B.second.NumUnits++;
157 }
158 }
159 }
160
161 // A SchedWrite may specify a number of cycles in which a resource group
162 // is reserved. For example (on target x86; cpu Haswell):
163 //
164 // SchedWriteRes<[HWPort0, HWPort1, HWPort01]> {
165 // let ReleaseAtCycles = [2, 2, 3];
166 // }
167 //
168 // This means:
169 // Resource units HWPort0 and HWPort1 are both used for 2cy.
170 // Resource group HWPort01 is the union of HWPort0 and HWPort1.
171 // Since this write touches both HWPort0 and HWPort1 for 2cy, HWPort01
172 // will not be usable for 2 entire cycles from instruction issue.
173 //
174 // On top of those 2cy, SchedWriteRes explicitly specifies an extra latency
175 // of 3 cycles for HWPort01. This tool assumes that the 3cy latency is an
176 // extra delay on top of the 2 cycles latency.
177 // During those extra cycles, HWPort01 is not usable by other instructions.
178 for (ResourcePlusCycles &RPC : ID.Resources) {
179 if (llvm::popcount(RPC.first) > 1 && !RPC.second.isReserved()) {
180 // Remove the leading 1 from the resource group mask.
181 uint64_t Mask = RPC.first ^ llvm::bit_floor(RPC.first);
182 uint64_t MaxResourceUnits = llvm::popcount(Mask);
183 if (RPC.second.NumUnits > (unsigned)llvm::popcount(Mask)) {
184 RPC.second.setReserved();
185 RPC.second.NumUnits = MaxResourceUnits;
186 }
187 }
188 }
189
190 // Identify extra buffers that are consumed through super resources.
191 for (const auto &SR : SuperResources) {
192 for (unsigned I = 1, E = NumProcResources; I < E; ++I) {
193 if (SM.getResourceBufferSize(I) == -1)
194 continue;
195
196 uint64_t Mask = ProcResourceMasks[I];
197 if (Mask != SR.first && ((Mask & SR.first) == SR.first))
198 Buffers.setBit(getResourceStateIndex(Mask));
199 }
200 }
201
202 ID.UsedBuffers = Buffers.getZExtValue();
203 ID.UsedProcResUnits = UsedResourceUnits;
204 ID.UsedProcResGroups = UsedResourceGroups;
205
206 LLVM_DEBUG({
207 for (const std::pair<uint64_t, ResourceUsage> &R : ID.Resources)
208 dbgs() << "\t\tResource Mask=" << format_hex(R.first, 16) << ", "
209 << "Reserved=" << R.second.isReserved() << ", "
210 << "#Units=" << R.second.NumUnits << ", "
211 << "cy=" << R.second.size() << '\n';
212 uint64_t BufferIDs = ID.UsedBuffers;
213 while (BufferIDs) {
214 uint64_t Current = BufferIDs & (-BufferIDs);
215 dbgs() << "\t\tBuffer Mask=" << format_hex(Current, 16) << '\n';
216 BufferIDs ^= Current;
217 }
218 dbgs() << "\t\t Used Units=" << format_hex(ID.UsedProcResUnits, 16) << '\n';
219 dbgs() << "\t\tUsed Groups=" << format_hex(ID.UsedProcResGroups, 16)
220 << '\n';
221 dbgs() << "\t\tHasPartiallyOverlappingGroups="
222 << ID.HasPartiallyOverlappingGroups << '\n';
223 });
224}
225
226static void computeMaxLatency(InstrDesc &ID, const MCSchedClassDesc &SCDesc,
227 const MCSubtargetInfo &STI, unsigned CallLatency,
228 bool IsCall) {
229 if (IsCall) {
230 // We cannot estimate how long this call will take.
231 // Artificially set an arbitrarily high latency.
232 ID.MaxLatency = CallLatency;
233 return;
234 }
235
237 // If latency is unknown, then conservatively assume the MaxLatency set for
238 // calls.
240}
241
242static Error verifyOperands(const MCInstrDesc &MCDesc, const MCInst &MCI) {
243 // Count register definitions, and skip non register operands in the process.
244 unsigned I, E;
245 unsigned NumExplicitDefs = MCDesc.getNumDefs();
246 for (I = 0, E = MCI.getNumOperands(); NumExplicitDefs && I < E; ++I) {
247 const MCOperand &Op = MCI.getOperand(I);
248 if (Op.isReg())
249 --NumExplicitDefs;
250 }
251
252 if (NumExplicitDefs) {
254 "Expected more register operand definitions.", MCI);
255 }
256
257 if (MCDesc.hasOptionalDef()) {
258 // Always assume that the optional definition is the last operand.
259 const MCOperand &Op = MCI.getOperand(MCDesc.getNumOperands() - 1);
260 if (I == MCI.getNumOperands() || !Op.isReg()) {
261 std::string Message =
262 "expected a register operand for an optional definition. Instruction "
263 "has not been correctly analyzed.";
264 return make_error<InstructionError<MCInst>>(Message, MCI);
265 }
266 }
267
268 return ErrorSuccess();
269}
270
271void InstrBuilder::populateWrites(InstrDesc &ID, const MCInst &MCI,
272 unsigned SchedClassID) {
273 const MCInstrDesc &MCDesc = MCII.get(MCI.getOpcode());
274 const MCSchedModel &SM = STI.getSchedModel();
275 const MCSchedClassDesc &SCDesc = *SM.getSchedClassDesc(SchedClassID);
276
277 // Assumptions made by this algorithm:
278 // 1. The number of explicit and implicit register definitions in a MCInst
279 // matches the number of explicit and implicit definitions according to
280 // the opcode descriptor (MCInstrDesc).
281 // 2. Uses start at index #(MCDesc.getNumDefs()).
282 // 3. There can only be a single optional register definition, an it is
283 // either the last operand of the sequence (excluding extra operands
284 // contributed by variadic opcodes) or one of the explicit register
285 // definitions. The latter occurs for some Thumb1 instructions.
286 //
287 // These assumptions work quite well for most out-of-order in-tree targets
288 // like x86. This is mainly because the vast majority of instructions is
289 // expanded to MCInst using a straightforward lowering logic that preserves
290 // the ordering of the operands.
291 //
292 // About assumption 1.
293 // The algorithm allows non-register operands between register operand
294 // definitions. This helps to handle some special ARM instructions with
295 // implicit operand increment (-mtriple=armv7):
296 //
297 // vld1.32 {d18, d19}, [r1]! @ <MCInst #1463 VLD1q32wb_fixed
298 // @ <MCOperand Reg:59>
299 // @ <MCOperand Imm:0> (!!)
300 // @ <MCOperand Reg:67>
301 // @ <MCOperand Imm:0>
302 // @ <MCOperand Imm:14>
303 // @ <MCOperand Reg:0>>
304 //
305 // MCDesc reports:
306 // 6 explicit operands.
307 // 1 optional definition
308 // 2 explicit definitions (!!)
309 //
310 // The presence of an 'Imm' operand between the two register definitions
311 // breaks the assumption that "register definitions are always at the
312 // beginning of the operand sequence".
313 //
314 // To workaround this issue, this algorithm ignores (i.e. skips) any
315 // non-register operands between register definitions. The optional
316 // definition is still at index #(NumOperands-1).
317 //
318 // According to assumption 2. register reads start at #(NumExplicitDefs-1).
319 // That means, register R1 from the example is both read and written.
320 unsigned NumExplicitDefs = MCDesc.getNumDefs();
321 unsigned NumImplicitDefs = MCDesc.implicit_defs().size();
322 unsigned NumWriteLatencyEntries = SCDesc.NumWriteLatencyEntries;
323 unsigned TotalDefs = NumExplicitDefs + NumImplicitDefs;
324 if (MCDesc.hasOptionalDef())
325 TotalDefs++;
326
327 unsigned NumVariadicOps = MCI.getNumOperands() - MCDesc.getNumOperands();
328 ID.Writes.resize(TotalDefs + NumVariadicOps);
329 // Iterate over the operands list, and skip non-register or constant register
330 // operands. The first NumExplicitDefs register operands are expected to be
331 // register definitions.
332 unsigned CurrentDef = 0;
333 unsigned OptionalDefIdx = MCDesc.getNumOperands() - 1;
334 unsigned i = 0;
335 for (; i < MCI.getNumOperands() && CurrentDef < NumExplicitDefs; ++i) {
336 const MCOperand &Op = MCI.getOperand(i);
337 if (!Op.isReg())
338 continue;
339
340 if (MCDesc.operands()[CurrentDef].isOptionalDef()) {
341 OptionalDefIdx = CurrentDef++;
342 continue;
343 }
344
345 WriteDescriptor &Write = ID.Writes[CurrentDef];
346 Write.OpIndex = i;
347 if (CurrentDef < NumWriteLatencyEntries) {
348 const MCWriteLatencyEntry &WLE =
349 *STI.getWriteLatencyEntry(&SCDesc, CurrentDef);
350 // Conservatively default to MaxLatency.
351 Write.Latency =
352 WLE.Cycles < 0 ? ID.MaxLatency : static_cast<unsigned>(WLE.Cycles);
353 Write.SClassOrWriteResourceID = WLE.WriteResourceID;
354 } else {
355 // Assign a default latency for this write.
356 Write.Latency = ID.MaxLatency;
357 Write.SClassOrWriteResourceID = 0;
358 }
359 Write.IsOptionalDef = false;
360 LLVM_DEBUG({
361 dbgs() << "\t\t[Def] OpIdx=" << Write.OpIndex
362 << ", Latency=" << Write.Latency
363 << ", WriteResourceID=" << Write.SClassOrWriteResourceID << '\n';
364 });
365 CurrentDef++;
366 }
367
368 assert(CurrentDef == NumExplicitDefs &&
369 "Expected more register operand definitions.");
370 for (CurrentDef = 0; CurrentDef < NumImplicitDefs; ++CurrentDef) {
371 unsigned Index = NumExplicitDefs + CurrentDef;
372 WriteDescriptor &Write = ID.Writes[Index];
373 Write.OpIndex = ~CurrentDef;
374 Write.RegisterID = MCDesc.implicit_defs()[CurrentDef];
375 if (Index < NumWriteLatencyEntries) {
376 const MCWriteLatencyEntry &WLE =
377 *STI.getWriteLatencyEntry(&SCDesc, Index);
378 // Conservatively default to MaxLatency.
379 Write.Latency =
380 WLE.Cycles < 0 ? ID.MaxLatency : static_cast<unsigned>(WLE.Cycles);
381 Write.SClassOrWriteResourceID = WLE.WriteResourceID;
382 } else {
383 // Assign a default latency for this write.
384 Write.Latency = ID.MaxLatency;
385 Write.SClassOrWriteResourceID = 0;
386 }
387
388 Write.IsOptionalDef = false;
389 assert(Write.RegisterID != 0 && "Expected a valid phys register!");
390 LLVM_DEBUG({
391 dbgs() << "\t\t[Def][I] OpIdx=" << ~Write.OpIndex
392 << ", PhysReg=" << MRI.getName(Write.RegisterID)
393 << ", Latency=" << Write.Latency
394 << ", WriteResourceID=" << Write.SClassOrWriteResourceID << '\n';
395 });
396 }
397
398 if (MCDesc.hasOptionalDef()) {
399 WriteDescriptor &Write = ID.Writes[NumExplicitDefs + NumImplicitDefs];
400 Write.OpIndex = OptionalDefIdx;
401 // Assign a default latency for this write.
402 Write.Latency = ID.MaxLatency;
403 Write.SClassOrWriteResourceID = 0;
404 Write.IsOptionalDef = true;
405 LLVM_DEBUG({
406 dbgs() << "\t\t[Def][O] OpIdx=" << Write.OpIndex
407 << ", Latency=" << Write.Latency
408 << ", WriteResourceID=" << Write.SClassOrWriteResourceID << '\n';
409 });
410 }
411
412 if (!NumVariadicOps)
413 return;
414
415 bool AssumeUsesOnly = !MCDesc.variadicOpsAreDefs();
416 CurrentDef = NumExplicitDefs + NumImplicitDefs + MCDesc.hasOptionalDef();
417 for (unsigned I = 0, OpIndex = MCDesc.getNumOperands();
418 I < NumVariadicOps && !AssumeUsesOnly; ++I, ++OpIndex) {
419 const MCOperand &Op = MCI.getOperand(OpIndex);
420 if (!Op.isReg())
421 continue;
422
423 WriteDescriptor &Write = ID.Writes[CurrentDef];
424 Write.OpIndex = OpIndex;
425 // Assign a default latency for this write.
426 Write.Latency = ID.MaxLatency;
427 Write.SClassOrWriteResourceID = 0;
428 Write.IsOptionalDef = false;
429 ++CurrentDef;
430 LLVM_DEBUG({
431 dbgs() << "\t\t[Def][V] OpIdx=" << Write.OpIndex
432 << ", Latency=" << Write.Latency
433 << ", WriteResourceID=" << Write.SClassOrWriteResourceID << '\n';
434 });
435 }
436
437 ID.Writes.resize(CurrentDef);
438}
439
440void InstrBuilder::populateReads(InstrDesc &ID, const MCInst &MCI,
441 unsigned SchedClassID) {
442 const MCInstrDesc &MCDesc = MCII.get(MCI.getOpcode());
443 unsigned NumExplicitUses = MCDesc.getNumOperands() - MCDesc.getNumDefs();
444 unsigned NumImplicitUses = MCDesc.implicit_uses().size();
445 // Remove the optional definition.
446 if (MCDesc.hasOptionalDef())
447 --NumExplicitUses;
448 unsigned NumVariadicOps = MCI.getNumOperands() - MCDesc.getNumOperands();
449 unsigned TotalUses = NumExplicitUses + NumImplicitUses + NumVariadicOps;
450 ID.Reads.resize(TotalUses);
451 unsigned CurrentUse = 0;
452 for (unsigned I = 0, OpIndex = MCDesc.getNumDefs(); I < NumExplicitUses;
453 ++I, ++OpIndex) {
454 const MCOperand &Op = MCI.getOperand(OpIndex);
455 if (!Op.isReg())
456 continue;
457
458 ReadDescriptor &Read = ID.Reads[CurrentUse];
459 Read.OpIndex = OpIndex;
460 Read.UseIndex = I;
461 Read.SchedClassID = SchedClassID;
462 ++CurrentUse;
463 LLVM_DEBUG(dbgs() << "\t\t[Use] OpIdx=" << Read.OpIndex
464 << ", UseIndex=" << Read.UseIndex << '\n');
465 }
466
467 // For the purpose of ReadAdvance, implicit uses come directly after explicit
468 // uses. The "UseIndex" must be updated according to that implicit layout.
469 for (unsigned I = 0; I < NumImplicitUses; ++I) {
470 ReadDescriptor &Read = ID.Reads[CurrentUse + I];
471 Read.OpIndex = ~I;
472 Read.UseIndex = NumExplicitUses + I;
473 Read.RegisterID = MCDesc.implicit_uses()[I];
474 Read.SchedClassID = SchedClassID;
475 LLVM_DEBUG(dbgs() << "\t\t[Use][I] OpIdx=" << ~Read.OpIndex
476 << ", UseIndex=" << Read.UseIndex << ", RegisterID="
477 << MRI.getName(Read.RegisterID) << '\n');
478 }
479
480 CurrentUse += NumImplicitUses;
481
482 bool AssumeDefsOnly = MCDesc.variadicOpsAreDefs();
483 for (unsigned I = 0, OpIndex = MCDesc.getNumOperands();
484 I < NumVariadicOps && !AssumeDefsOnly; ++I, ++OpIndex) {
485 const MCOperand &Op = MCI.getOperand(OpIndex);
486 if (!Op.isReg())
487 continue;
488
489 ReadDescriptor &Read = ID.Reads[CurrentUse];
490 Read.OpIndex = OpIndex;
491 Read.UseIndex = NumExplicitUses + NumImplicitUses + I;
492 Read.SchedClassID = SchedClassID;
493 ++CurrentUse;
494 LLVM_DEBUG(dbgs() << "\t\t[Use][V] OpIdx=" << Read.OpIndex
495 << ", UseIndex=" << Read.UseIndex << '\n');
496 }
497
498 ID.Reads.resize(CurrentUse);
499}
500
502 hash_code TypeHash = hash_combine(MCO.isReg(), MCO.isImm(), MCO.isSFPImm(),
503 MCO.isDFPImm(), MCO.isExpr(), MCO.isInst());
504 if (MCO.isReg())
505 return hash_combine(TypeHash, MCO.getReg());
506
507 return TypeHash;
508}
509
511 hash_code InstructionHash = hash_combine(MCI.getOpcode(), MCI.getFlags());
512 for (unsigned I = 0; I < MCI.getNumOperands(); ++I) {
513 InstructionHash =
514 hash_combine(InstructionHash, hashMCOperand(MCI.getOperand(I)));
515 }
516 return InstructionHash;
517}
518
519Error InstrBuilder::verifyInstrDesc(const InstrDesc &ID,
520 const MCInst &MCI) const {
521 if (ID.NumMicroOps != 0)
522 return ErrorSuccess();
523
524 bool UsesBuffers = ID.UsedBuffers;
525 bool UsesResources = !ID.Resources.empty();
526 if (!UsesBuffers && !UsesResources)
527 return ErrorSuccess();
528
529 // FIXME: see PR44797. We should revisit these checks and possibly move them
530 // in CodeGenSchedule.cpp.
531 StringRef Message = "found an inconsistent instruction that decodes to zero "
532 "opcodes and that consumes scheduler resources.";
533 return make_error<InstructionError<MCInst>>(std::string(Message), MCI);
534}
535
536Expected<unsigned> InstrBuilder::getVariantSchedClassID(const MCInst &MCI,
537 unsigned SchedClassID) {
538 const MCSchedModel &SM = STI.getSchedModel();
539 unsigned CPUID = SM.getProcessorID();
540 while (SchedClassID && SM.getSchedClassDesc(SchedClassID)->isVariant())
541 SchedClassID =
542 STI.resolveVariantSchedClass(SchedClassID, &MCI, &MCII, CPUID);
543
544 if (!SchedClassID) {
546 "unable to resolve scheduling class for write variant.", MCI);
547 }
548
549 return SchedClassID;
550}
551
552Expected<const InstrDesc &>
553InstrBuilder::createInstrDescImpl(const MCInst &MCI,
554 const SmallVector<Instrument *> &IVec) {
555 assert(STI.getSchedModel().hasInstrSchedModel() &&
556 "Itineraries are not yet supported!");
557
558 // Obtain the instruction descriptor from the opcode.
559 unsigned Opcode = MCI.getOpcode();
560 const MCInstrDesc &MCDesc = MCII.get(Opcode);
561 const MCSchedModel &SM = STI.getSchedModel();
562
563 // Then obtain the scheduling class information from the instruction.
564 // Allow InstrumentManager to override and use a different SchedClassID
565 unsigned SchedClassID = IM.getSchedClassID(MCII, MCI, IVec);
566 bool IsVariant = SM.getSchedClassDesc(SchedClassID)->isVariant();
567
568 // Try to solve variant scheduling classes.
569 if (IsVariant) {
570 Expected<unsigned> VariantSchedClassIDOrErr =
571 getVariantSchedClassID(MCI, SchedClassID);
572 if (!VariantSchedClassIDOrErr) {
573 return VariantSchedClassIDOrErr.takeError();
574 }
575
576 SchedClassID = *VariantSchedClassIDOrErr;
577 }
578
579 // Check if this instruction is supported. Otherwise, report an error.
580 const MCSchedClassDesc &SCDesc = *SM.getSchedClassDesc(SchedClassID);
581 if (SCDesc.NumMicroOps == MCSchedClassDesc::InvalidNumMicroOps) {
583 "found an unsupported instruction in the input assembly sequence", MCI);
584 }
585
586 LLVM_DEBUG(dbgs() << "\n\t\tOpcode Name= " << MCII.getName(Opcode) << '\n');
587 LLVM_DEBUG(dbgs() << "\t\tSchedClassID=" << SchedClassID << '\n');
588 LLVM_DEBUG(dbgs() << "\t\tOpcode=" << Opcode << '\n');
589
590 // Create a new empty descriptor.
591 std::unique_ptr<InstrDesc> ID = std::make_unique<InstrDesc>();
592 ID->NumMicroOps = SCDesc.NumMicroOps;
593 ID->SchedClassID = SchedClassID;
594
595 bool IsCall = MCIA->isCall(MCI);
596 if (IsCall && FirstCallInst) {
597 // We don't correctly model calls.
598 WithColor::warning() << "found a call in the input assembly sequence.\n";
599 WithColor::note() << "call instructions are not correctly modeled. "
600 << "Assume a latency of " << CallLatency << "cy.\n";
601 FirstCallInst = false;
602 }
603
604 if (MCIA->isReturn(MCI) && FirstReturnInst) {
605 WithColor::warning() << "found a return instruction in the input"
606 << " assembly sequence.\n";
607 WithColor::note() << "program counter updates are ignored.\n";
608 FirstReturnInst = false;
609 }
610
611 initializeUsedResources(*ID, SCDesc, STI, ProcResourceMasks);
612 computeMaxLatency(*ID, SCDesc, STI, CallLatency, IsCall);
613
614 if (Error Err = verifyOperands(MCDesc, MCI))
615 return std::move(Err);
616
617 populateWrites(*ID, MCI, SchedClassID);
618 populateReads(*ID, MCI, SchedClassID);
619
620 LLVM_DEBUG(dbgs() << "\t\tMaxLatency=" << ID->MaxLatency << '\n');
621 LLVM_DEBUG(dbgs() << "\t\tNumMicroOps=" << ID->NumMicroOps << '\n');
622
623 // Validation check on the instruction descriptor.
624 if (Error Err = verifyInstrDesc(*ID, MCI))
625 return std::move(Err);
626
627 // Now add the new descriptor.
628
629 if (IM.canCustomize(IVec)) {
630 IM.customize(IVec, *ID);
631 return *CustomDescriptors.emplace_back(std::move(ID));
632 }
633
634 bool IsVariadic = MCDesc.isVariadic();
635 if ((ID->IsRecyclable = !IsVariadic && !IsVariant)) {
636 auto DKey = std::make_pair(MCI.getOpcode(), SchedClassID);
637 return *(Descriptors[DKey] = std::move(ID));
638 }
639
640 auto VDKey = std::make_pair(hashMCInst(MCI), SchedClassID);
641 assert(
642 !VariantDescriptors.contains(VDKey) &&
643 "Expected VariantDescriptors to not already have a value for this key.");
644 return *(VariantDescriptors[VDKey] = std::move(ID));
645}
646
647Expected<const InstrDesc &>
648InstrBuilder::getOrCreateInstrDesc(const MCInst &MCI,
649 const SmallVector<Instrument *> &IVec) {
650 // Cache lookup using SchedClassID from Instrumentation
651 unsigned SchedClassID = IM.getSchedClassID(MCII, MCI, IVec);
652
653 auto DKey = std::make_pair(MCI.getOpcode(), SchedClassID);
654 if (Descriptors.find_as(DKey) != Descriptors.end())
655 return *Descriptors[DKey];
656
657 Expected<unsigned> VariantSchedClassIDOrErr =
658 getVariantSchedClassID(MCI, SchedClassID);
659 if (!VariantSchedClassIDOrErr) {
660 return VariantSchedClassIDOrErr.takeError();
661 }
662
663 SchedClassID = *VariantSchedClassIDOrErr;
664
665 auto VDKey = std::make_pair(hashMCInst(MCI), SchedClassID);
666 auto It = VariantDescriptors.find(VDKey);
667 if (It != VariantDescriptors.end())
668 return *It->second;
669
670 return createInstrDescImpl(MCI, IVec);
671}
672
673STATISTIC(NumVariantInst, "Number of MCInsts that doesn't have static Desc");
674
677 const SmallVector<Instrument *> &IVec) {
678 Expected<const InstrDesc &> DescOrErr = IM.canCustomize(IVec)
679 ? createInstrDescImpl(MCI, IVec)
680 : getOrCreateInstrDesc(MCI, IVec);
681 if (!DescOrErr)
682 return DescOrErr.takeError();
683 const InstrDesc &D = *DescOrErr;
684 Instruction *NewIS = nullptr;
685 std::unique_ptr<Instruction> CreatedIS;
686 bool IsInstRecycled = false;
687
688 if (!D.IsRecyclable)
689 ++NumVariantInst;
690
691 if (D.IsRecyclable && InstRecycleCB) {
692 if (auto *I = InstRecycleCB(D)) {
693 NewIS = I;
694 NewIS->reset();
695 IsInstRecycled = true;
696 }
697 }
698 if (!IsInstRecycled) {
699 CreatedIS = std::make_unique<Instruction>(D, MCI.getOpcode());
700 NewIS = CreatedIS.get();
701 }
702
703 const MCInstrDesc &MCDesc = MCII.get(MCI.getOpcode());
704 const MCSchedClassDesc &SCDesc =
705 *STI.getSchedModel().getSchedClassDesc(D.SchedClassID);
706
707 NewIS->setMayLoad(MCDesc.mayLoad());
708 NewIS->setMayStore(MCDesc.mayStore());
710 NewIS->setBeginGroup(SCDesc.BeginGroup);
711 NewIS->setEndGroup(SCDesc.EndGroup);
712 NewIS->setRetireOOO(SCDesc.RetireOOO);
713
714 // Check if this is a dependency breaking instruction.
715 APInt Mask;
716
717 bool IsZeroIdiom = false;
718 bool IsDepBreaking = false;
719 if (MCIA) {
720 unsigned ProcID = STI.getSchedModel().getProcessorID();
721 IsZeroIdiom = MCIA->isZeroIdiom(MCI, Mask, ProcID);
722 IsDepBreaking =
723 IsZeroIdiom || MCIA->isDependencyBreaking(MCI, Mask, ProcID);
724 if (MCIA->isOptimizableRegisterMove(MCI, ProcID))
725 NewIS->setOptimizableMove();
726 }
727
728 // Initialize Reads first.
729 MCPhysReg RegID = 0;
730 size_t Idx = 0U;
731 for (const ReadDescriptor &RD : D.Reads) {
732 if (!RD.isImplicitRead()) {
733 // explicit read.
734 const MCOperand &Op = MCI.getOperand(RD.OpIndex);
735 // Skip non-register operands.
736 if (!Op.isReg())
737 continue;
738 // Skip constant register operands.
739 if (MRI.isConstant(Op.getReg()))
740 continue;
741 RegID = Op.getReg().id();
742 } else {
743 // Implicit read.
744 RegID = RD.RegisterID;
745 }
746
747 // Skip invalid register operands.
748 if (!RegID)
749 continue;
750
751 // Okay, this is a register operand. Create a ReadState for it.
752 ReadState *RS = nullptr;
753 if (IsInstRecycled && Idx < NewIS->getUses().size()) {
754 NewIS->getUses()[Idx] = ReadState(RD, RegID);
755 RS = &NewIS->getUses()[Idx++];
756 } else {
757 NewIS->getUses().emplace_back(RD, RegID);
758 RS = &NewIS->getUses().back();
759 ++Idx;
760 }
761
762 if (IsDepBreaking) {
763 // A mask of all zeroes means: explicit input operands are not
764 // independent.
765 if (Mask.isZero()) {
766 if (!RD.isImplicitRead())
767 RS->setIndependentFromDef();
768 } else {
769 // Check if this register operand is independent according to `Mask`.
770 // Note that Mask may not have enough bits to describe all explicit and
771 // implicit input operands. If this register operand doesn't have a
772 // corresponding bit in Mask, then conservatively assume that it is
773 // dependent.
774 if (Mask.getBitWidth() > RD.UseIndex) {
775 // Okay. This map describe register use `RD.UseIndex`.
776 if (Mask[RD.UseIndex])
777 RS->setIndependentFromDef();
778 }
779 }
780 }
781 }
782 if (IsInstRecycled && Idx < NewIS->getUses().size())
783 NewIS->getUses().pop_back_n(NewIS->getUses().size() - Idx);
784
785 // Early exit if there are no writes.
786 if (D.Writes.empty()) {
787 if (IsInstRecycled)
789 else
790 return std::move(CreatedIS);
791 }
792
793 // Track register writes that implicitly clear the upper portion of the
794 // underlying super-registers using an APInt.
795 APInt WriteMask(D.Writes.size(), 0);
796
797 // Now query the MCInstrAnalysis object to obtain information about which
798 // register writes implicitly clear the upper portion of a super-register.
799 if (MCIA)
800 MCIA->clearsSuperRegisters(MRI, MCI, WriteMask);
801
802 // Initialize writes.
803 unsigned WriteIndex = 0;
804 Idx = 0U;
805 for (const WriteDescriptor &WD : D.Writes) {
806 RegID = WD.isImplicitWrite() ? WD.RegisterID
807 : MCI.getOperand(WD.OpIndex).getReg().id();
808 // Check if this is a optional definition that references NoReg or a write
809 // to a constant register.
810 if ((WD.IsOptionalDef && !RegID) || MRI.isConstant(RegID)) {
811 ++WriteIndex;
812 continue;
813 }
814
815 assert(RegID && "Expected a valid register ID!");
816 if (IsInstRecycled && Idx < NewIS->getDefs().size()) {
817 NewIS->getDefs()[Idx++] =
818 WriteState(WD, RegID,
819 /* ClearsSuperRegs */ WriteMask[WriteIndex],
820 /* WritesZero */ IsZeroIdiom);
821 } else {
822 NewIS->getDefs().emplace_back(WD, RegID,
823 /* ClearsSuperRegs */ WriteMask[WriteIndex],
824 /* WritesZero */ IsZeroIdiom);
825 ++Idx;
826 }
827 ++WriteIndex;
828 }
829 if (IsInstRecycled && Idx < NewIS->getDefs().size())
830 NewIS->getDefs().pop_back_n(NewIS->getDefs().size() - Idx);
831
832 if (IsInstRecycled)
834 else
835 return std::move(CreatedIS);
836}
837} // namespace mca
838} // namespace llvm
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
This file implements a class to represent arbitrary precision integral constant values and operations...
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_EXPORT_TEMPLATE
Definition Compiler.h:217
This file defines the DenseMap class.
#define im(i)
A builder class for instructions that are statically analyzed by llvm-mca.
#define I(x, y, z)
Definition MD5.cpp:57
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Subclass of Error for the sole purpose of identifying the success path in the type system.
Definition Error.h:334
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
Tagged union holding either a T or a Error.
Definition Error.h:485
Error takeError()
Take ownership of the stored error.
Definition Error.h:612
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
unsigned getNumOperands() const
Definition MCInst.h:212
unsigned getFlags() const
Definition MCInst.h:205
unsigned getOpcode() const
Definition MCInst.h:202
const MCOperand & getOperand(unsigned i) const
Definition MCInst.h:210
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
bool mayStore() const
Return true if this instruction could possibly modify memory.
bool mayLoad() const
Return true if this instruction could possibly read memory.
bool hasOptionalDef() const
Set if this instruction has an optional definition, e.g.
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
bool variadicOpsAreDefs() const
Return true if variadic operands of this instruction are definitions.
ArrayRef< MCPhysReg > implicit_defs() const
Return a list of registers that are potentially written by any instance of this machine instruction.
bool hasUnmodeledSideEffects() const
Return true if this instruction has side effects that are not modeled by other flags.
Interface to description of machine instruction set.
Definition MCInstrInfo.h:27
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
Instances of this class represent operands of the MCInst class.
Definition MCInst.h:40
bool isSFPImm() const
Definition MCInst.h:67
bool isImm() const
Definition MCInst.h:66
bool isInst() const
Definition MCInst.h:70
bool isReg() const
Definition MCInst.h:65
MCRegister getReg() const
Returns the register number.
Definition MCInst.h:73
bool isDFPImm() const
Definition MCInst.h:68
bool isExpr() const
Definition MCInst.h:69
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
constexpr unsigned id() const
Definition MCRegister.h:82
Generic base class for all target subtargets.
virtual unsigned resolveVariantSchedClass(unsigned SchedClass, const MCInst *MI, const MCInstrInfo *MCII, unsigned CPUID) const
Resolve a variant scheduling class for the given MCInst and CPU.
const MCWriteLatencyEntry * getWriteLatencyEntry(const MCSchedClassDesc *SC, unsigned DefIdx) const
const MCWriteProcResEntry * getWriteProcResBegin(const MCSchedClassDesc *SC) const
Return an iterator at the first process resource consumed by the given scheduling class.
const MCSchedModel & getSchedModel() const
Get the machine model for this subtarget's CPU.
reference emplace_back(ArgTypes &&... Args)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static LLVM_ABI raw_ostream & warning()
Convenience method for printing "warning: " to stderr.
Definition WithColor.cpp:86
static LLVM_ABI raw_ostream & note()
Convenience method for printing "note: " to stderr.
Definition WithColor.cpp:88
An opaque object representing a hash code.
Definition Hashing.h:77
A sequence of cycles.
LLVM_ABI Expected< std::unique_ptr< Instruction > > createInstruction(const MCInst &MCI, const SmallVector< Instrument * > &IVec)
void setEndGroup(bool newVal)
void setRetireOOO(bool newVal)
SmallVectorImpl< WriteState > & getDefs()
void setBeginGroup(bool newVal)
SmallVectorImpl< ReadState > & getUses()
void setHasSideEffects(bool newVal)
void setMayStore(bool newVal)
void setMayLoad(bool newVal)
An instruction propagated through the simulated instruction pipeline.
LLVM_ABI void reset()
This class allows targets to optionally customize the logic that resolves scheduling class IDs.
Tracks register operand latency in cycles.
static LLVM_ABI char ID
Tracks uses of a register definition (e.g.
Helper functions used by various pipeline components.
This namespace contains all of the command line option processing machinery.
Definition MCSchedule.h:35
static void computeMaxLatency(InstrDesc &ID, const MCSchedClassDesc &SCDesc, const MCSubtargetInfo &STI, unsigned CallLatency, bool IsCall)
char InstructionError< T >::ID
Definition Support.h:44
hash_code hashMCInst(const MCInst &MCI)
static void initializeUsedResources(InstrDesc &ID, const MCSchedClassDesc &SCDesc, const MCSubtargetInfo &STI, ArrayRef< uint64_t > ProcResourceMasks)
hash_code hashMCOperand(const MCOperand &MCO)
LLVM_ABI void computeProcResourceMasks(const MCSchedModel &SM, MutableArrayRef< uint64_t > Masks)
Populates vector Masks with processor resource masks.
Definition Support.cpp:41
unsigned getResourceStateIndex(uint64_t Mask)
Definition Support.h:107
static Error verifyOperands(const MCInstrDesc &MCDesc, const MCInst &MCI)
This is an optimization pass for GlobalISel generic memory operations.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
FormattedNumber format_hex(uint64_t N, unsigned Width, bool Upper=false)
format_hex - Output N as a fixed width hexadecimal.
Definition Format.h:164
Error make_error(ArgTs &&... Args)
Make a Error instance representing failure using the given error info type.
Definition Error.h:340
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
Definition Hashing.h:307
@ TypeHash
Token ID based on allocated type hash.
Definition AllocToken.h:32
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
Define a kind of processor resource that will be modeled by the scheduler.
Definition MCSchedule.h:42
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Definition MCSchedule.h:129
static const unsigned short InvalidNumMicroOps
Definition MCSchedule.h:130
Machine model for scheduling, bundling, and heuristics.
Definition MCSchedule.h:273
const MCSchedClassDesc * getSchedClassDesc(unsigned SchedClassIdx) const
Definition MCSchedule.h:381
unsigned getProcessorID() const
Definition MCSchedule.h:352
unsigned getNumProcResourceKinds() const
Definition MCSchedule.h:370
static LLVM_ABI int computeInstrLatency(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Returns the latency value for the scheduling class.
const MCProcResourceDesc * getProcResource(unsigned ProcResourceIdx) const
Definition MCSchedule.h:374
LLVM_ABI int getResourceBufferSize(unsigned ProcResourceIdx) const
Return the buffer size of the resource.
StringRef getSchedClassName(unsigned SchedClassIdx) const
Definition MCSchedule.h:388
Identify one of the processor resource kinds consumed by a particular scheduling class for the specif...
Definition MCSchedule.h:74
uint16_t ReleaseAtCycle
Cycle at which the resource will be released by an instruction, relatively to the cycle in which the ...
Definition MCSchedule.h:79
An instruction descriptor.
A register read descriptor.
Helper used by class InstrDesc to describe how hardware resources are used.
A register write descriptor.