From e5ea38a757a91840a51449d4af30d31da9236d3e Mon Sep 17 00:00:00 2001 From: Slava Egorov Date: Sat, 25 Feb 2023 22:03:08 +0000 Subject: [PATCH] Revert "[vm/compiler] Split ParallelMove codegen into scheduling and emission" This reverts commit 32b093379b354137221bc134eb0dc5fb3a474b1c. Reason for revert: various test failures across the board. Original change's description: > [vm/compiler] Split ParallelMove codegen into scheduling and emission > > This CL does not contain any changes to behaviour, but simply moves > ParallelMoveResolver to a separate file. Additionally instead of > immediately generating code we produce a move schedule which is > attached to the ParallelMoveInstr and later converted to the > native code. > > This refactoring prepares the code for subsequent improvements, e.g. > we want to rework how temporaries used by move resolution are > allocated: instead of pushing/poping them around every move that needs > them we will allocate space for them in spill area. > > Having ParallelMove scheduling separated from code emission also > allows to unit test it. > > TEST=ci > > Change-Id: If3f7a88836037a9812a85c1cfc2ef21a7fe15747 > Reviewed-on: https://dart-review.googlesource.com/c/sdk/+/284222 > Commit-Queue: Slava Egorov > Reviewed-by: Alexander Markov > Reviewed-by: Martin Kustermann Change-Id: I82952d024816327ca5f084a2185fa1ab566cfa82 No-Presubmit: true No-Tree-Checks: true No-Try: true Reviewed-on: https://dart-review.googlesource.com/c/sdk/+/285560 Auto-Submit: Slava Egorov Commit-Queue: Rubber Stamper Bot-Commit: Rubber Stamper --- .../compiler/backend/flow_graph_compiler.cc | 296 +++++++++++++- .../vm/compiler/backend/flow_graph_compiler.h | 106 +++++ .../backend/flow_graph_compiler_arm.cc | 57 ++- .../backend/flow_graph_compiler_arm64.cc | 57 ++- .../backend/flow_graph_compiler_ia32.cc | 57 ++- .../backend/flow_graph_compiler_riscv.cc | 57 ++- .../backend/flow_graph_compiler_x64.cc | 57 ++- runtime/vm/compiler/backend/il.cc | 11 +- runtime/vm/compiler/backend/il.h | 13 - runtime/vm/compiler/backend/il_arm.cc | 4 +- runtime/vm/compiler/backend/il_arm64.cc | 4 +- runtime/vm/compiler/backend/il_ia32.cc | 4 +- runtime/vm/compiler/backend/il_riscv.cc | 4 +- runtime/vm/compiler/backend/il_x64.cc | 4 +- runtime/vm/compiler/backend/linearscan.cc | 23 -- runtime/vm/compiler/backend/linearscan.h | 2 - .../backend/parallel_move_resolver.cc | 387 ------------------ .../compiler/backend/parallel_move_resolver.h | 157 ------- runtime/vm/compiler/compiler_sources.gni | 2 - runtime/vm/compiler/graph_intrinsifier.cc | 16 +- runtime/vm/globals.h | 2 +- 21 files changed, 600 insertions(+), 720 deletions(-) delete mode 100644 runtime/vm/compiler/backend/parallel_move_resolver.cc delete mode 100644 runtime/vm/compiler/backend/parallel_move_resolver.h diff --git a/runtime/vm/compiler/backend/flow_graph_compiler.cc b/runtime/vm/compiler/backend/flow_graph_compiler.cc index 5ad339341bf..e149b66a070 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler.cc +++ b/runtime/vm/compiler/backend/flow_graph_compiler.cc @@ -166,6 +166,7 @@ FlowGraphCompiler::FlowGraphCompiler( Class::ZoneHandle(isolate_group()->object_store()->int32x4_class())), list_class_(Class::ZoneHandle(Library::Handle(Library::CoreLibrary()) .LookupClass(Symbols::List()))), + parallel_move_resolver_(this), pending_deoptimization_env_(NULL), deopt_id_to_ic_data_(deopt_id_to_ic_data), edge_counters_array_(Array::ZoneHandle()) { @@ -736,22 +737,25 @@ void FlowGraphCompiler::VisitBlocks() { } EmitComment(instr); } - - BeginCodeSourceRange(instr->source()); - EmitInstructionPrologue(instr); - ASSERT(pending_deoptimization_env_ == NULL); - pending_deoptimization_env_ = instr->env(); - DEBUG_ONLY(current_instruction_ = instr); - instr->EmitNativeCode(this); - DEBUG_ONLY(current_instruction_ = nullptr); - pending_deoptimization_env_ = NULL; - if (IsPeephole(instr)) { - ASSERT(top_of_stack_ == nullptr); - top_of_stack_ = instr->AsDefinition(); + if (instr->IsParallelMove()) { + parallel_move_resolver_.EmitNativeCode(instr->AsParallelMove()); } else { - EmitInstructionEpilogue(instr); + BeginCodeSourceRange(instr->source()); + EmitInstructionPrologue(instr); + ASSERT(pending_deoptimization_env_ == NULL); + pending_deoptimization_env_ = instr->env(); + DEBUG_ONLY(current_instruction_ = instr); + instr->EmitNativeCode(this); + DEBUG_ONLY(current_instruction_ = nullptr); + pending_deoptimization_env_ = NULL; + if (IsPeephole(instr)) { + ASSERT(top_of_stack_ == nullptr); + top_of_stack_ = instr->AsDefinition(); + } else { + EmitInstructionEpilogue(instr); + } + EndCodeSourceRange(instr->source()); } - EndCodeSourceRange(instr->source()); #if defined(DEBUG) if (!is_optimizing()) { @@ -1850,6 +1854,270 @@ void FlowGraphCompiler::AllocateRegistersLocally(Instruction* instr) { } } +static uword RegMaskBit(Register reg) { + return ((reg) != kNoRegister) ? (1 << (reg)) : 0; +} + +ParallelMoveResolver::ParallelMoveResolver(FlowGraphCompiler* compiler) + : compiler_(compiler), moves_(32) {} + +void ParallelMoveResolver::EmitNativeCode(ParallelMoveInstr* parallel_move) { + ASSERT(moves_.is_empty()); + + // Build up a worklist of moves. + BuildInitialMoveList(parallel_move); + + const InstructionSource& move_source = InstructionSource( + TokenPosition::kParallelMove, parallel_move->inlining_id()); + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& move = *moves_[i]; + // Skip constants to perform them last. They don't block other moves + // and skipping such moves with register destinations keeps those + // registers free for the whole algorithm. + if (!move.IsEliminated() && !move.src().IsConstant()) { + PerformMove(move_source, i); + } + } + + // Perform the moves with constant sources. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& move = *moves_[i]; + if (!move.IsEliminated()) { + ASSERT(move.src().IsConstant()); + compiler_->BeginCodeSourceRange(move_source); + EmitMove(i); + compiler_->EndCodeSourceRange(move_source); + } + } + + moves_.Clear(); +} + +void ParallelMoveResolver::BuildInitialMoveList( + ParallelMoveInstr* parallel_move) { + // Perform a linear sweep of the moves to add them to the initial list of + // moves to perform, ignoring any move that is redundant (the source is + // the same as the destination, the destination is ignored and + // unallocated, or the move was already eliminated). + for (int i = 0; i < parallel_move->NumMoves(); i++) { + MoveOperands* move = parallel_move->MoveOperandsAt(i); + if (!move->IsRedundant()) moves_.Add(move); + } +} + +void ParallelMoveResolver::PerformMove(const InstructionSource& source, + int index) { + // Each call to this function performs a move and deletes it from the move + // graph. We first recursively perform any move blocking this one. We + // mark a move as "pending" on entry to PerformMove in order to detect + // cycles in the move graph. We use operand swaps to resolve cycles, + // which means that a call to PerformMove could change any source operand + // in the move graph. + + ASSERT(!moves_[index]->IsPending()); + ASSERT(!moves_[index]->IsRedundant()); + + // Clear this move's destination to indicate a pending move. The actual + // destination is saved in a stack-allocated local. Recursion may allow + // multiple moves to be pending. + ASSERT(!moves_[index]->src().IsInvalid()); + Location destination = moves_[index]->MarkPending(); + + // Perform a depth-first traversal of the move graph to resolve + // dependencies. Any unperformed, unpending move with a source the same + // as this one's destination blocks this one so recursively perform all + // such moves. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(destination) && !other_move.IsPending()) { + // Though PerformMove can change any source operand in the move graph, + // this call cannot create a blocking move via a swap (this loop does + // not miss any). Assume there is a non-blocking move with source A + // and this move is blocked on source B and there is a swap of A and + // B. Then A and B must be involved in the same cycle (or they would + // not be swapped). Since this move's destination is B and there is + // only a single incoming edge to an operand, this move must also be + // involved in the same cycle. In that case, the blocking move will + // be created but will be "pending" when we return from PerformMove. + PerformMove(source, i); + } + } + + // We are about to resolve this move and don't need it marked as + // pending, so restore its destination. + moves_[index]->ClearPending(destination); + + // This move's source may have changed due to swaps to resolve cycles and + // so it may now be the last move in the cycle. If so remove it. + if (moves_[index]->src().Equals(destination)) { + moves_[index]->Eliminate(); + return; + } + + // The move may be blocked on a (at most one) pending move, in which case + // we have a cycle. Search for such a blocking move and perform a swap to + // resolve it. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(destination)) { + ASSERT(other_move.IsPending()); + compiler_->BeginCodeSourceRange(source); + EmitSwap(index); + compiler_->EndCodeSourceRange(source); + return; + } + } + + // This move is not blocked. + compiler_->BeginCodeSourceRange(source); + EmitMove(index); + compiler_->EndCodeSourceRange(source); +} + +void ParallelMoveResolver::EmitMove(int index) { + MoveOperands* const move = moves_[index]; + const Location dst = move->dest(); + if (dst.IsStackSlot() || dst.IsDoubleStackSlot()) { + ASSERT((dst.base_reg() != FPREG) || + ((-compiler::target::frame_layout.VariableIndexForFrameSlot( + dst.stack_index())) < compiler_->StackSize())); + } + const Location src = move->src(); + ParallelMoveResolver::TemporaryAllocator temp(this, /*blocked=*/kNoRegister); + compiler_->EmitMove(dst, src, &temp); +#if defined(DEBUG) + // Allocating a scratch register here may cause stack spilling. Neither the + // source nor destination register should be SP-relative in that case. + for (const Location& loc : {dst, src}) { + ASSERT(!temp.DidAllocateTemporary() || !loc.HasStackIndex() || + loc.base_reg() != SPREG); + } +#endif + move->Eliminate(); +} + +bool ParallelMoveResolver::IsScratchLocation(Location loc) { + for (int i = 0; i < moves_.length(); ++i) { + if (moves_[i]->Blocks(loc)) { + return false; + } + } + + for (int i = 0; i < moves_.length(); ++i) { + if (moves_[i]->dest().Equals(loc)) { + return true; + } + } + + return false; +} + +intptr_t ParallelMoveResolver::AllocateScratchRegister( + Location::Kind kind, + uword blocked_mask, + intptr_t first_free_register, + intptr_t last_free_register, + bool* spilled) { + COMPILE_ASSERT(static_cast(sizeof(blocked_mask)) * kBitsPerByte >= + kNumberOfFpuRegisters); + COMPILE_ASSERT(static_cast(sizeof(blocked_mask)) * kBitsPerByte >= + kNumberOfCpuRegisters); + intptr_t scratch = -1; + for (intptr_t reg = first_free_register; reg <= last_free_register; reg++) { + if ((((1 << reg) & blocked_mask) == 0) && + IsScratchLocation(Location::MachineRegisterLocation(kind, reg))) { + scratch = reg; + break; + } + } + + if (scratch == -1) { + *spilled = true; + for (intptr_t reg = first_free_register; reg <= last_free_register; reg++) { + if (((1 << reg) & blocked_mask) == 0) { + scratch = reg; + break; + } + } + } else { + *spilled = false; + } + + return scratch; +} + +ParallelMoveResolver::ScratchFpuRegisterScope::ScratchFpuRegisterScope( + ParallelMoveResolver* resolver, + FpuRegister blocked) + : resolver_(resolver), reg_(kNoFpuRegister), spilled_(false) { + COMPILE_ASSERT(FpuTMP != kNoFpuRegister); + uword blocked_mask = + ((blocked != kNoFpuRegister) ? 1 << blocked : 0) | 1 << FpuTMP; + reg_ = static_cast(resolver_->AllocateScratchRegister( + Location::kFpuRegister, blocked_mask, 0, kNumberOfFpuRegisters - 1, + &spilled_)); + + if (spilled_) { + resolver->SpillFpuScratch(reg_); + } +} + +ParallelMoveResolver::ScratchFpuRegisterScope::~ScratchFpuRegisterScope() { + if (spilled_) { + resolver_->RestoreFpuScratch(reg_); + } +} + +ParallelMoveResolver::TemporaryAllocator::TemporaryAllocator( + ParallelMoveResolver* resolver, + Register blocked) + : resolver_(resolver), + blocked_(blocked), + reg_(kNoRegister), + spilled_(false) {} + +Register ParallelMoveResolver::TemporaryAllocator::AllocateTemporary() { + ASSERT(reg_ == kNoRegister); + + uword blocked_mask = RegMaskBit(blocked_) | kReservedCpuRegisters; + if (resolver_->compiler_->intrinsic_mode()) { + // Block additional registers that must be preserved for intrinsics. + blocked_mask |= RegMaskBit(ARGS_DESC_REG); +#if !defined(TARGET_ARCH_IA32) + // Need to preserve CODE_REG to be able to store the PC marker + // and load the pool pointer. + blocked_mask |= RegMaskBit(CODE_REG); +#endif + } + reg_ = static_cast( + resolver_->AllocateScratchRegister(Location::kRegister, blocked_mask, 0, + kNumberOfCpuRegisters - 1, &spilled_)); + + if (spilled_) { + resolver_->SpillScratch(reg_); + } + + DEBUG_ONLY(allocated_ = true;) + return reg_; +} + +void ParallelMoveResolver::TemporaryAllocator::ReleaseTemporary() { + if (spilled_) { + resolver_->RestoreScratch(reg_); + } + reg_ = kNoRegister; +} + +ParallelMoveResolver::ScratchRegisterScope::ScratchRegisterScope( + ParallelMoveResolver* resolver, + Register blocked) + : allocator_(resolver, blocked) { + reg_ = allocator_.AllocateTemporary(); +} + +ParallelMoveResolver::ScratchRegisterScope::~ScratchRegisterScope() { + allocator_.ReleaseTemporary(); +} const ICData* FlowGraphCompiler::GetOrAddInstanceCallICData( intptr_t deopt_id, diff --git a/runtime/vm/compiler/backend/flow_graph_compiler.h b/runtime/vm/compiler/backend/flow_graph_compiler.h index 6be75b07184..96918f2285a 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler.h +++ b/runtime/vm/compiler/backend/flow_graph_compiler.h @@ -63,6 +63,107 @@ class NoTemporaryAllocator : public TemporaryRegisterAllocator { void ReleaseTemporary() override { UNREACHABLE(); } }; +class ParallelMoveResolver : public ValueObject { + public: + explicit ParallelMoveResolver(FlowGraphCompiler* compiler); + + // Resolve a set of parallel moves, emitting assembler instructions. + void EmitNativeCode(ParallelMoveInstr* parallel_move); + + private: + class ScratchFpuRegisterScope : public ValueObject { + public: + ScratchFpuRegisterScope(ParallelMoveResolver* resolver, + FpuRegister blocked); + ~ScratchFpuRegisterScope(); + + FpuRegister reg() const { return reg_; } + + private: + ParallelMoveResolver* resolver_; + FpuRegister reg_; + bool spilled_; + }; + + class TemporaryAllocator : public TemporaryRegisterAllocator { + public: + TemporaryAllocator(ParallelMoveResolver* resolver, Register blocked); + + Register AllocateTemporary() override; + void ReleaseTemporary() override; + DEBUG_ONLY(bool DidAllocateTemporary() { return allocated_; }) + + virtual ~TemporaryAllocator() { ASSERT(reg_ == kNoRegister); } + + private: + ParallelMoveResolver* const resolver_; + const Register blocked_; + Register reg_; + bool spilled_; + DEBUG_ONLY(bool allocated_ = false); + }; + + class ScratchRegisterScope : public ValueObject { + public: + ScratchRegisterScope(ParallelMoveResolver* resolver, Register blocked); + ~ScratchRegisterScope(); + + Register reg() const { return reg_; } + + private: + TemporaryAllocator allocator_; + Register reg_; + }; + + bool IsScratchLocation(Location loc); + intptr_t AllocateScratchRegister(Location::Kind kind, + uword blocked_mask, + intptr_t first_free_register, + intptr_t last_free_register, + bool* spilled); + + void SpillScratch(Register reg); + void RestoreScratch(Register reg); + void SpillFpuScratch(FpuRegister reg); + void RestoreFpuScratch(FpuRegister reg); + + // friend class ScratchXmmRegisterScope; + + // Build the initial list of moves. + void BuildInitialMoveList(ParallelMoveInstr* parallel_move); + + // Perform the move at the moves_ index in question (possibly requiring + // other moves to satisfy dependencies). + void PerformMove(const InstructionSource& source, int index); + + // Emit a move and remove it from the move graph. + void EmitMove(int index); + + // Execute a move by emitting a swap of two operands. The move from + // source to destination is removed from the move graph. + void EmitSwap(int index); + + // Verify the move list before performing moves. + void Verify(); + + // Helpers for non-trivial source-destination combinations that cannot + // be handled by a single instruction. + void MoveMemoryToMemory(const compiler::Address& dst, + const compiler::Address& src); + void Exchange(Register reg, const compiler::Address& mem); + void Exchange(const compiler::Address& mem1, const compiler::Address& mem2); + void Exchange(Register reg, Register base_reg, intptr_t stack_offset); + void Exchange(Register base_reg1, + intptr_t stack_offset1, + Register base_reg2, + intptr_t stack_offset2); + + FlowGraphCompiler* compiler_; + + // List of moves not yet resolved. + GrowableArray moves_; +}; + // Used for describing a deoptimization point after call (lazy deoptimization). // For deoptimization before instruction use class CompilerDeoptInfoWithStub. class CompilerDeoptInfo : public ZoneAllocated { @@ -444,6 +545,9 @@ class FlowGraphCompiler : public ValueObject { bool ForceSlowPathForStackOverflow() const; const GrowableArray& block_info() const { return block_info_; } + ParallelMoveResolver* parallel_move_resolver() { + return ¶llel_move_resolver_; + } void StatsBegin(Instruction* instr) { if (stats_ != NULL) stats_->Begin(instr); @@ -1186,6 +1290,8 @@ class FlowGraphCompiler : public ValueObject { const Class& int32x4_class_; const Class& list_class_; + ParallelMoveResolver parallel_move_resolver_; + // Currently instructions generate deopt stubs internally by // calling AddDeoptStub. To communicate deoptimization environment // that should be used when deoptimizing we store it in this variable. diff --git a/runtime/vm/compiler/backend/flow_graph_compiler_arm.cc b/runtime/vm/compiler/backend/flow_graph_compiler_arm.cc index eb9f0310419..1f511774c87 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler_arm.cc +++ b/runtime/vm/compiler/backend/flow_graph_compiler_arm.cc @@ -10,7 +10,6 @@ #include "vm/compiler/api/type_check_mode.h" #include "vm/compiler/backend/il_printer.h" #include "vm/compiler/backend/locations.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/compiler/jit/compiler.h" #include "vm/cpu.h" #include "vm/dart_entry.h" @@ -1102,9 +1101,10 @@ void FlowGraphCompiler::LoadBSSEntry(BSS::Relocation relocation, #undef __ #define __ compiler_->assembler()-> -void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { - const Location source = move.src(); - const Location destination = move.dest(); +void ParallelMoveResolver::EmitSwap(int index) { + MoveOperands* move = moves_[index]; + const Location source = move->src(); + const Location destination = move->dest(); if (source.IsRegister() && destination.IsRegister()) { ASSERT(source.reg() != IP); @@ -1183,39 +1183,56 @@ void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { } else { UNREACHABLE(); } + + // The swap of source and destination has executed a move from source to + // destination. + move->Eliminate(); + + // Any unperformed (including pending) move with a source of either + // this move's source or destination needs to have their source + // changed to reflect the state of affairs after the swap. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(source)) { + moves_[i]->set_src(destination); + } else if (other_move.Blocks(destination)) { + moves_[i]->set_src(source); + } + } } -void ParallelMoveEmitter::MoveMemoryToMemory(const compiler::Address& dst, - const compiler::Address& src) { +void ParallelMoveResolver::MoveMemoryToMemory(const compiler::Address& dst, + const compiler::Address& src) { UNREACHABLE(); } // Do not call or implement this function. Instead, use the form below that // uses an offset from the frame pointer instead of an Address. -void ParallelMoveEmitter::Exchange(Register reg, const compiler::Address& mem) { +void ParallelMoveResolver::Exchange(Register reg, + const compiler::Address& mem) { UNREACHABLE(); } // Do not call or implement this function. Instead, use the form below that // uses offsets from the frame pointer instead of Addresses. -void ParallelMoveEmitter::Exchange(const compiler::Address& mem1, - const compiler::Address& mem2) { +void ParallelMoveResolver::Exchange(const compiler::Address& mem1, + const compiler::Address& mem2) { UNREACHABLE(); } -void ParallelMoveEmitter::Exchange(Register reg, - Register base_reg, - intptr_t stack_offset) { +void ParallelMoveResolver::Exchange(Register reg, + Register base_reg, + intptr_t stack_offset) { ScratchRegisterScope tmp(this, reg); __ mov(tmp.reg(), compiler::Operand(reg)); __ LoadFromOffset(reg, base_reg, stack_offset); __ StoreToOffset(tmp.reg(), base_reg, stack_offset); } -void ParallelMoveEmitter::Exchange(Register base_reg1, - intptr_t stack_offset1, - Register base_reg2, - intptr_t stack_offset2) { +void ParallelMoveResolver::Exchange(Register base_reg1, + intptr_t stack_offset1, + Register base_reg2, + intptr_t stack_offset2) { ScratchRegisterScope tmp1(this, kNoRegister); ScratchRegisterScope tmp2(this, tmp1.reg()); __ LoadFromOffset(tmp1.reg(), base_reg1, stack_offset1); @@ -1224,19 +1241,19 @@ void ParallelMoveEmitter::Exchange(Register base_reg1, __ StoreToOffset(tmp2.reg(), base_reg1, stack_offset1); } -void ParallelMoveEmitter::SpillScratch(Register reg) { +void ParallelMoveResolver::SpillScratch(Register reg) { __ Push(reg); } -void ParallelMoveEmitter::RestoreScratch(Register reg) { +void ParallelMoveResolver::RestoreScratch(Register reg) { __ Pop(reg); } -void ParallelMoveEmitter::SpillFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::SpillFpuScratch(FpuRegister reg) { __ PushQuad(reg); } -void ParallelMoveEmitter::RestoreFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::RestoreFpuScratch(FpuRegister reg) { __ PopQuad(reg); } diff --git a/runtime/vm/compiler/backend/flow_graph_compiler_arm64.cc b/runtime/vm/compiler/backend/flow_graph_compiler_arm64.cc index d5e954aae47..a18eb702635 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler_arm64.cc +++ b/runtime/vm/compiler/backend/flow_graph_compiler_arm64.cc @@ -10,7 +10,6 @@ #include "vm/compiler/api/type_check_mode.h" #include "vm/compiler/backend/il_printer.h" #include "vm/compiler/backend/locations.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/compiler/jit/compiler.h" #include "vm/cpu.h" #include "vm/dart_entry.h" @@ -1077,9 +1076,10 @@ void FlowGraphCompiler::LoadBSSEntry(BSS::Relocation relocation, #undef __ #define __ compiler_->assembler()-> -void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { - const Location source = move.src(); - const Location destination = move.dest(); +void ParallelMoveResolver::EmitSwap(int index) { + MoveOperands* move = moves_[index]; + const Location source = move->src(); + const Location destination = move->dest(); if (source.IsRegister() && destination.IsRegister()) { ASSERT(source.reg() != TMP); @@ -1146,39 +1146,56 @@ void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { } else { UNREACHABLE(); } + + // The swap of source and destination has executed a move from source to + // destination. + move->Eliminate(); + + // Any unperformed (including pending) move with a source of either + // this move's source or destination needs to have their source + // changed to reflect the state of affairs after the swap. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(source)) { + moves_[i]->set_src(destination); + } else if (other_move.Blocks(destination)) { + moves_[i]->set_src(source); + } + } } -void ParallelMoveEmitter::MoveMemoryToMemory(const compiler::Address& dst, - const compiler::Address& src) { +void ParallelMoveResolver::MoveMemoryToMemory(const compiler::Address& dst, + const compiler::Address& src) { UNREACHABLE(); } // Do not call or implement this function. Instead, use the form below that // uses an offset from the frame pointer instead of an Address. -void ParallelMoveEmitter::Exchange(Register reg, const compiler::Address& mem) { +void ParallelMoveResolver::Exchange(Register reg, + const compiler::Address& mem) { UNREACHABLE(); } // Do not call or implement this function. Instead, use the form below that // uses offsets from the frame pointer instead of Addresses. -void ParallelMoveEmitter::Exchange(const compiler::Address& mem1, - const compiler::Address& mem2) { +void ParallelMoveResolver::Exchange(const compiler::Address& mem1, + const compiler::Address& mem2) { UNREACHABLE(); } -void ParallelMoveEmitter::Exchange(Register reg, - Register base_reg, - intptr_t stack_offset) { +void ParallelMoveResolver::Exchange(Register reg, + Register base_reg, + intptr_t stack_offset) { ScratchRegisterScope tmp(this, reg); __ mov(tmp.reg(), reg); __ LoadFromOffset(reg, base_reg, stack_offset); __ StoreToOffset(tmp.reg(), base_reg, stack_offset); } -void ParallelMoveEmitter::Exchange(Register base_reg1, - intptr_t stack_offset1, - Register base_reg2, - intptr_t stack_offset2) { +void ParallelMoveResolver::Exchange(Register base_reg1, + intptr_t stack_offset1, + Register base_reg2, + intptr_t stack_offset2) { ScratchRegisterScope tmp1(this, kNoRegister); ScratchRegisterScope tmp2(this, tmp1.reg()); __ LoadFromOffset(tmp1.reg(), base_reg1, stack_offset1); @@ -1187,19 +1204,19 @@ void ParallelMoveEmitter::Exchange(Register base_reg1, __ StoreToOffset(tmp2.reg(), base_reg1, stack_offset1); } -void ParallelMoveEmitter::SpillScratch(Register reg) { +void ParallelMoveResolver::SpillScratch(Register reg) { __ Push(reg); } -void ParallelMoveEmitter::RestoreScratch(Register reg) { +void ParallelMoveResolver::RestoreScratch(Register reg) { __ Pop(reg); } -void ParallelMoveEmitter::SpillFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::SpillFpuScratch(FpuRegister reg) { __ PushQuad(reg); } -void ParallelMoveEmitter::RestoreFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::RestoreFpuScratch(FpuRegister reg) { __ PopQuad(reg); } diff --git a/runtime/vm/compiler/backend/flow_graph_compiler_ia32.cc b/runtime/vm/compiler/backend/flow_graph_compiler_ia32.cc index 18dced18d97..f114ffb68cf 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler_ia32.cc +++ b/runtime/vm/compiler/backend/flow_graph_compiler_ia32.cc @@ -11,7 +11,6 @@ #include "vm/compiler/api/type_check_mode.h" #include "vm/compiler/backend/il_printer.h" #include "vm/compiler/backend/locations.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/compiler/frontend/flow_graph_builder.h" #include "vm/compiler/jit/compiler.h" #include "vm/cpu.h" @@ -1031,9 +1030,10 @@ void FlowGraphCompiler::EmitNativeMoveArchitecture( #undef __ #define __ compiler_->assembler()-> -void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { - const Location source = move.src(); - const Location destination = move.dest(); +void ParallelMoveResolver::EmitSwap(int index) { + MoveOperands* move = moves_[index]; + const Location source = move->src(); + const Location destination = move->dest(); if (source.IsRegister() && destination.IsRegister()) { __ xchgl(destination.reg(), source.reg()); @@ -1092,23 +1092,40 @@ void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { } else { UNREACHABLE(); } + + // The swap of source and destination has executed a move from source to + // destination. + move->Eliminate(); + + // Any unperformed (including pending) move with a source of either + // this move's source or destination needs to have their source + // changed to reflect the state of affairs after the swap. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(source)) { + moves_[i]->set_src(destination); + } else if (other_move.Blocks(destination)) { + moves_[i]->set_src(source); + } + } } -void ParallelMoveEmitter::MoveMemoryToMemory(const compiler::Address& dst, - const compiler::Address& src) { +void ParallelMoveResolver::MoveMemoryToMemory(const compiler::Address& dst, + const compiler::Address& src) { ScratchRegisterScope ensure_scratch(this, kNoRegister); __ MoveMemoryToMemory(dst, src, ensure_scratch.reg()); } -void ParallelMoveEmitter::Exchange(Register reg, const compiler::Address& mem) { +void ParallelMoveResolver::Exchange(Register reg, + const compiler::Address& mem) { ScratchRegisterScope ensure_scratch(this, reg); __ movl(ensure_scratch.reg(), mem); __ movl(mem, reg); __ movl(reg, ensure_scratch.reg()); } -void ParallelMoveEmitter::Exchange(const compiler::Address& mem1, - const compiler::Address& mem2) { +void ParallelMoveResolver::Exchange(const compiler::Address& mem1, + const compiler::Address& mem2) { ScratchRegisterScope ensure_scratch1(this, kNoRegister); ScratchRegisterScope ensure_scratch2(this, ensure_scratch1.reg()); __ movl(ensure_scratch1.reg(), mem1); @@ -1117,33 +1134,33 @@ void ParallelMoveEmitter::Exchange(const compiler::Address& mem1, __ movl(mem1, ensure_scratch2.reg()); } -void ParallelMoveEmitter::Exchange(Register reg, - Register base_reg, - intptr_t stack_offset) { +void ParallelMoveResolver::Exchange(Register reg, + Register base_reg, + intptr_t stack_offset) { UNREACHABLE(); } -void ParallelMoveEmitter::Exchange(Register base_reg1, - intptr_t stack_offset1, - Register base_reg2, - intptr_t stack_offset2) { +void ParallelMoveResolver::Exchange(Register base_reg1, + intptr_t stack_offset1, + Register base_reg2, + intptr_t stack_offset2) { UNREACHABLE(); } -void ParallelMoveEmitter::SpillScratch(Register reg) { +void ParallelMoveResolver::SpillScratch(Register reg) { __ pushl(reg); } -void ParallelMoveEmitter::RestoreScratch(Register reg) { +void ParallelMoveResolver::RestoreScratch(Register reg) { __ popl(reg); } -void ParallelMoveEmitter::SpillFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::SpillFpuScratch(FpuRegister reg) { __ subl(ESP, compiler::Immediate(kFpuRegisterSize)); __ movups(compiler::Address(ESP, 0), reg); } -void ParallelMoveEmitter::RestoreFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::RestoreFpuScratch(FpuRegister reg) { __ movups(reg, compiler::Address(ESP, 0)); __ addl(ESP, compiler::Immediate(kFpuRegisterSize)); } diff --git a/runtime/vm/compiler/backend/flow_graph_compiler_riscv.cc b/runtime/vm/compiler/backend/flow_graph_compiler_riscv.cc index bea2c36337f..ca0e3c07bcb 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler_riscv.cc +++ b/runtime/vm/compiler/backend/flow_graph_compiler_riscv.cc @@ -10,7 +10,6 @@ #include "vm/compiler/api/type_check_mode.h" #include "vm/compiler/backend/il_printer.h" #include "vm/compiler/backend/locations.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/compiler/jit/compiler.h" #include "vm/cpu.h" #include "vm/dart_entry.h" @@ -1081,9 +1080,10 @@ void FlowGraphCompiler::LoadBSSEntry(BSS::Relocation relocation, #undef __ #define __ compiler_->assembler()-> -void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { - const Location source = move.src(); - const Location destination = move.dest(); +void ParallelMoveResolver::EmitSwap(int index) { + MoveOperands* move = moves_[index]; + const Location source = move->src(); + const Location destination = move->dest(); if (source.IsRegister() && destination.IsRegister()) { ASSERT(source.reg() != TMP); @@ -1122,38 +1122,55 @@ void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { } else { UNREACHABLE(); } + + // The swap of source and destination has executed a move from source to + // destination. + move->Eliminate(); + + // Any unperformed (including pending) move with a source of either + // this move's source or destination needs to have their source + // changed to reflect the state of affairs after the swap. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(source)) { + moves_[i]->set_src(destination); + } else if (other_move.Blocks(destination)) { + moves_[i]->set_src(source); + } + } } -void ParallelMoveEmitter::MoveMemoryToMemory(const compiler::Address& dst, - const compiler::Address& src) { +void ParallelMoveResolver::MoveMemoryToMemory(const compiler::Address& dst, + const compiler::Address& src) { UNREACHABLE(); } // Do not call or implement this function. Instead, use the form below that // uses an offset from the frame pointer instead of an Address. -void ParallelMoveEmitter::Exchange(Register reg, const compiler::Address& mem) { +void ParallelMoveResolver::Exchange(Register reg, + const compiler::Address& mem) { UNREACHABLE(); } // Do not call or implement this function. Instead, use the form below that // uses offsets from the frame pointer instead of Addresses. -void ParallelMoveEmitter::Exchange(const compiler::Address& mem1, - const compiler::Address& mem2) { +void ParallelMoveResolver::Exchange(const compiler::Address& mem1, + const compiler::Address& mem2) { UNREACHABLE(); } -void ParallelMoveEmitter::Exchange(Register reg, - Register base_reg, - intptr_t stack_offset) { +void ParallelMoveResolver::Exchange(Register reg, + Register base_reg, + intptr_t stack_offset) { __ mv(TMP, reg); __ LoadFromOffset(reg, base_reg, stack_offset); __ StoreToOffset(TMP, base_reg, stack_offset); } -void ParallelMoveEmitter::Exchange(Register base_reg1, - intptr_t stack_offset1, - Register base_reg2, - intptr_t stack_offset2) { +void ParallelMoveResolver::Exchange(Register base_reg1, + intptr_t stack_offset1, + Register base_reg2, + intptr_t stack_offset2) { ScratchRegisterScope tmp1(this, kNoRegister); ScratchRegisterScope tmp2(this, tmp1.reg()); __ LoadFromOffset(tmp1.reg(), base_reg1, stack_offset1); @@ -1162,20 +1179,20 @@ void ParallelMoveEmitter::Exchange(Register base_reg1, __ StoreToOffset(tmp2.reg(), base_reg1, stack_offset1); } -void ParallelMoveEmitter::SpillScratch(Register reg) { +void ParallelMoveResolver::SpillScratch(Register reg) { __ PushRegister(reg); } -void ParallelMoveEmitter::RestoreScratch(Register reg) { +void ParallelMoveResolver::RestoreScratch(Register reg) { __ PopRegister(reg); } -void ParallelMoveEmitter::SpillFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::SpillFpuScratch(FpuRegister reg) { __ subi(SP, SP, sizeof(double)); __ fsd(reg, compiler::Address(SP, 0)); } -void ParallelMoveEmitter::RestoreFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::RestoreFpuScratch(FpuRegister reg) { __ fld(reg, compiler::Address(SP, 0)); __ addi(SP, SP, sizeof(double)); } diff --git a/runtime/vm/compiler/backend/flow_graph_compiler_x64.cc b/runtime/vm/compiler/backend/flow_graph_compiler_x64.cc index 49796459190..d58491307c6 100644 --- a/runtime/vm/compiler/backend/flow_graph_compiler_x64.cc +++ b/runtime/vm/compiler/backend/flow_graph_compiler_x64.cc @@ -10,7 +10,6 @@ #include "vm/compiler/api/type_check_mode.h" #include "vm/compiler/backend/il_printer.h" #include "vm/compiler/backend/locations.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/compiler/ffi/native_location.h" #include "vm/compiler/jit/compiler.h" #include "vm/dart_entry.h" @@ -1069,9 +1068,10 @@ void FlowGraphCompiler::LoadBSSEntry(BSS::Relocation relocation, #undef __ #define __ compiler_->assembler()-> -void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { - const Location source = move.src(); - const Location destination = move.dest(); +void ParallelMoveResolver::EmitSwap(int index) { + MoveOperands* move = moves_[index]; + const Location source = move->src(); + const Location destination = move->dest(); if (source.IsRegister() && destination.IsRegister()) { __ xchgq(destination.reg(), source.reg()); @@ -1130,49 +1130,66 @@ void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) { } else { UNREACHABLE(); } + + // The swap of source and destination has executed a move from source to + // destination. + move->Eliminate(); + + // Any unperformed (including pending) move with a source of either + // this move's source or destination needs to have their source + // changed to reflect the state of affairs after the swap. + for (int i = 0; i < moves_.length(); ++i) { + const MoveOperands& other_move = *moves_[i]; + if (other_move.Blocks(source)) { + moves_[i]->set_src(destination); + } else if (other_move.Blocks(destination)) { + moves_[i]->set_src(source); + } + } } -void ParallelMoveEmitter::MoveMemoryToMemory(const compiler::Address& dst, - const compiler::Address& src) { +void ParallelMoveResolver::MoveMemoryToMemory(const compiler::Address& dst, + const compiler::Address& src) { __ MoveMemoryToMemory(dst, src); } -void ParallelMoveEmitter::Exchange(Register reg, const compiler::Address& mem) { +void ParallelMoveResolver::Exchange(Register reg, + const compiler::Address& mem) { __ Exchange(reg, mem); } -void ParallelMoveEmitter::Exchange(const compiler::Address& mem1, - const compiler::Address& mem2) { +void ParallelMoveResolver::Exchange(const compiler::Address& mem1, + const compiler::Address& mem2) { __ Exchange(mem1, mem2); } -void ParallelMoveEmitter::Exchange(Register reg, - Register base_reg, - intptr_t stack_offset) { +void ParallelMoveResolver::Exchange(Register reg, + Register base_reg, + intptr_t stack_offset) { UNREACHABLE(); } -void ParallelMoveEmitter::Exchange(Register base_reg1, - intptr_t stack_offset1, - Register base_reg2, - intptr_t stack_offset2) { +void ParallelMoveResolver::Exchange(Register base_reg1, + intptr_t stack_offset1, + Register base_reg2, + intptr_t stack_offset2) { UNREACHABLE(); } -void ParallelMoveEmitter::SpillScratch(Register reg) { +void ParallelMoveResolver::SpillScratch(Register reg) { __ pushq(reg); } -void ParallelMoveEmitter::RestoreScratch(Register reg) { +void ParallelMoveResolver::RestoreScratch(Register reg) { __ popq(reg); } -void ParallelMoveEmitter::SpillFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::SpillFpuScratch(FpuRegister reg) { __ AddImmediate(RSP, compiler::Immediate(-kFpuRegisterSize)); __ movups(compiler::Address(RSP, 0), reg); } -void ParallelMoveEmitter::RestoreFpuScratch(FpuRegister reg) { +void ParallelMoveResolver::RestoreFpuScratch(FpuRegister reg) { __ movups(reg, compiler::Address(RSP, 0)); __ AddImmediate(RSP, compiler::Immediate(kFpuRegisterSize)); } diff --git a/runtime/vm/compiler/backend/il.cc b/runtime/vm/compiler/backend/il.cc index 0340e758746..92cde47925c 100644 --- a/runtime/vm/compiler/backend/il.cc +++ b/runtime/vm/compiler/backend/il.cc @@ -16,7 +16,6 @@ #include "vm/compiler/backend/locations.h" #include "vm/compiler/backend/locations_helpers.h" #include "vm/compiler/backend/loops.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/compiler/backend/range_analysis.h" #include "vm/compiler/ffi/frame_rebase.h" #include "vm/compiler/ffi/marshaller.h" @@ -4036,7 +4035,7 @@ void JoinEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { InstructionSource()); } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } } @@ -4066,7 +4065,7 @@ void TargetEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { if (compiler::Assembler::EmittingComments()) { compiler->EmitComment(parallel_move()); } - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } } @@ -4140,7 +4139,7 @@ void FunctionEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { if (compiler::Assembler::EmittingComments()) { compiler->EmitComment(parallel_move()); } - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } } @@ -4236,7 +4235,7 @@ void OsrEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { if (compiler::Assembler::EmittingComments()) { compiler->EmitComment(parallel_move()); } - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } } @@ -4681,7 +4680,7 @@ LocationSummary* ParallelMoveInstr::MakeLocationSummary(Zone* zone, } void ParallelMoveInstr::EmitNativeCode(FlowGraphCompiler* compiler) { - ParallelMoveEmitter(compiler, this).EmitNativeCode(); + UNREACHABLE(); } LocationSummary* ConstraintInstr::MakeLocationSummary(Zone* zone, diff --git a/runtime/vm/compiler/backend/il.h b/runtime/vm/compiler/backend/il.h index 88c48a08e1b..9ce94ba9a77 100644 --- a/runtime/vm/compiler/backend/il.h +++ b/runtime/vm/compiler/backend/il.h @@ -59,7 +59,6 @@ class Instruction; class InstructionVisitor; class LocalVariable; class LoopInfo; -class MoveSchedule; class ParsedFunction; class Range; class RangeAnalysis; @@ -1480,8 +1479,6 @@ class TemplateInstruction class MoveOperands : public ZoneAllocated { public: MoveOperands(Location dest, Location src) : dest_(dest), src_(src) {} - MoveOperands(const MoveOperands& other) - : dest_(other.dest_), src_(other.src_) {} MoveOperands& operator=(const MoveOperands& other) { dest_ = other.dest_; @@ -1571,22 +1568,12 @@ class ParallelMoveInstr : public TemplateInstruction<0, NoThrow> { return TokenPosition::kParallelMove; } - const MoveSchedule& move_schedule() const { - ASSERT(move_schedule_ != nullptr); - return *move_schedule_; - } - - void set_move_schedule(const MoveSchedule& schedule) { - move_schedule_ = &schedule; - } - PRINT_TO_SUPPORT DECLARE_EMPTY_SERIALIZATION(ParallelMoveInstr, TemplateInstruction) DECLARE_EXTRA_SERIALIZATION private: GrowableArray moves_; // Elements cannot be null. - const MoveSchedule* move_schedule_ = nullptr; DISALLOW_COPY_AND_ASSIGN(ParallelMoveInstr); }; diff --git a/runtime/vm/compiler/backend/il_arm.cc b/runtime/vm/compiler/backend/il_arm.cc index ed5d945a241..739e1703c27 100644 --- a/runtime/vm/compiler/backend/il_arm.cc +++ b/runtime/vm/compiler/backend/il_arm.cc @@ -3187,7 +3187,7 @@ void CatchBlockEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { } } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // Restore SP from FP as we are coming from a throw and the code for @@ -7126,7 +7126,7 @@ void GotoInstr::EmitNativeCode(FlowGraphCompiler* compiler) { InstructionSource()); } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // We can fall through if the successor is the next block in the list. diff --git a/runtime/vm/compiler/backend/il_arm64.cc b/runtime/vm/compiler/backend/il_arm64.cc index 237327a2853..72e5a33b309 100644 --- a/runtime/vm/compiler/backend/il_arm64.cc +++ b/runtime/vm/compiler/backend/il_arm64.cc @@ -2847,7 +2847,7 @@ void CatchBlockEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { } } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // Restore SP from FP as we are coming from a throw and the code for @@ -6218,7 +6218,7 @@ void GotoInstr::EmitNativeCode(FlowGraphCompiler* compiler) { InstructionSource()); } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // We can fall through if the successor is the next block in the list. diff --git a/runtime/vm/compiler/backend/il_ia32.cc b/runtime/vm/compiler/backend/il_ia32.cc index e63955570ec..4b42108d17e 100644 --- a/runtime/vm/compiler/backend/il_ia32.cc +++ b/runtime/vm/compiler/backend/il_ia32.cc @@ -2474,7 +2474,7 @@ void CatchBlockEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { } } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // Restore ESP from EBP as we are coming from a throw and the code for @@ -6264,7 +6264,7 @@ void GotoInstr::EmitNativeCode(FlowGraphCompiler* compiler) { InstructionSource()); } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // We can fall through if the successor is the next block in the list. diff --git a/runtime/vm/compiler/backend/il_riscv.cc b/runtime/vm/compiler/backend/il_riscv.cc index c1dab4c79a6..c55624d05aa 100644 --- a/runtime/vm/compiler/backend/il_riscv.cc +++ b/runtime/vm/compiler/backend/il_riscv.cc @@ -3128,7 +3128,7 @@ void CatchBlockEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { } } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // Restore SP from FP as we are coming from a throw and the code for @@ -7243,7 +7243,7 @@ void GotoInstr::EmitNativeCode(FlowGraphCompiler* compiler) { InstructionSource()); } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // We can fall through if the successor is the next block in the list. diff --git a/runtime/vm/compiler/backend/il_x64.cc b/runtime/vm/compiler/backend/il_x64.cc index a3e9da8103b..68dd257f8eb 100644 --- a/runtime/vm/compiler/backend/il_x64.cc +++ b/runtime/vm/compiler/backend/il_x64.cc @@ -2894,7 +2894,7 @@ void CatchBlockEntryInstr::EmitNativeCode(FlowGraphCompiler* compiler) { } } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // Restore RSP from RBP as we are coming from a throw and the code for @@ -6583,7 +6583,7 @@ void GotoInstr::EmitNativeCode(FlowGraphCompiler* compiler) { InstructionSource()); } if (HasParallelMove()) { - parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode(parallel_move()); } // We can fall through if the successor is the next block in the list. diff --git a/runtime/vm/compiler/backend/linearscan.cc b/runtime/vm/compiler/backend/linearscan.cc index 8cf52b93f77..1632b440282 100644 --- a/runtime/vm/compiler/backend/linearscan.cc +++ b/runtime/vm/compiler/backend/linearscan.cc @@ -10,7 +10,6 @@ #include "vm/compiler/backend/il.h" #include "vm/compiler/backend/il_printer.h" #include "vm/compiler/backend/loops.h" -#include "vm/compiler/backend/parallel_move_resolver.h" #include "vm/log.h" #include "vm/parser.h" #include "vm/stack_frame.h" @@ -3312,26 +3311,6 @@ void FlowGraphAllocator::AllocateOutgoingArguments() { } } -void FlowGraphAllocator::ScheduleParallelMoves() { - ParallelMoveResolver resolver; - - for (auto block : flow_graph_.reverse_postorder()) { - if (block->HasParallelMove()) { - resolver.Resolve(block->parallel_move()); - } - for (auto instruction : block->instructions()) { - if (auto move = instruction->AsParallelMove()) { - resolver.Resolve(move); - } - } - if (auto goto_instr = block->last_instruction()->AsGoto()) { - if (goto_instr->HasParallelMove()) { - resolver.Resolve(goto_instr->parallel_move()); - } - } - } -} - void FlowGraphAllocator::AllocateRegisters() { CollectRepresentations(); @@ -3396,8 +3375,6 @@ void FlowGraphAllocator::AllocateRegisters() { ResolveControlFlow(); - ScheduleParallelMoves(); - if (FLAG_print_ssa_liveranges && CompilerState::ShouldTrace()) { const Function& function = flow_graph_.function(); diff --git a/runtime/vm/compiler/backend/linearscan.h b/runtime/vm/compiler/backend/linearscan.h index 62f859c4da4..42844822a87 100644 --- a/runtime/vm/compiler/backend/linearscan.h +++ b/runtime/vm/compiler/backend/linearscan.h @@ -181,8 +181,6 @@ class FlowGraphAllocator : public ValueObject { // Connect split siblings over non-linear control flow edges. void ResolveControlFlow(); - void ScheduleParallelMoves(); - // Returns true if the target location is the spill slot for the given range. bool TargetLocationIsSpillSlot(LiveRange* range, Location target); diff --git a/runtime/vm/compiler/backend/parallel_move_resolver.cc b/runtime/vm/compiler/backend/parallel_move_resolver.cc deleted file mode 100644 index 529cae3b641..00000000000 --- a/runtime/vm/compiler/backend/parallel_move_resolver.cc +++ /dev/null @@ -1,387 +0,0 @@ -// Copyright (c) 2023, the Dart project authors. Please see the AUTHORS file -// for details. All rights reserved. Use of this source code is governed by a -// BSD-style license that can be found in the LICENSE file. - -#include "vm/compiler/backend/parallel_move_resolver.h" - -namespace dart { - -// Simple dynamically allocated array of fixed length. -template -class FixedArray { - public: - static Subclass& Allocate(intptr_t length) { - static_assert(Utils::IsAligned(alignof(Subclass), alignof(Element))); - auto result = - reinterpret_cast(Thread::Current()->zone()->AllocUnsafe( - sizeof(Subclass) + length * sizeof(Element))); - return *new (result) Subclass(length); - } - - intptr_t length() const { return length_; } - - Element& operator[](intptr_t i) { - ASSERT(0 <= i && i < length_); - return data()[i]; - } - - const Element& operator[](intptr_t i) const { - ASSERT(0 <= i && i < length_); - return data()[i]; - } - - Element* data() { OPEN_ARRAY_START(Element, Element); } - const Element* data() const { OPEN_ARRAY_START(Element, Element); } - - Element* begin() { return data(); } - const Element* begin() const { return data(); } - - Element* end() { return data() + length_; } - const Element* end() const { return data() + length_; } - - protected: - explicit FixedArray(intptr_t length) : length_(length) {} - - private: - intptr_t length_; - - DISALLOW_COPY_AND_ASSIGN(FixedArray); -}; - -class MoveSchedule : public FixedArray { - public: - // Converts the given list of |ParallelMoveResolver::Op| operations - // into a |MoveSchedule| and filters out all |kNop| operations. - static const MoveSchedule& From( - const GrowableArray& ops) { - intptr_t count = 0; - for (const auto& op : ops) { - if (op.kind != ParallelMoveResolver::OpKind::kNop) count++; - } - - auto& result = FixedArray::Allocate(count); - intptr_t i = 0; - for (const auto& op : ops) { - if (op.kind != ParallelMoveResolver::OpKind::kNop) { - result[i++] = op; - } - } - return result; - } - - private: - friend class FixedArray; - - explicit MoveSchedule(intptr_t length) : FixedArray(length) {} - - DISALLOW_COPY_AND_ASSIGN(MoveSchedule); -}; - -static uword RegMaskBit(Register reg) { - return ((reg) != kNoRegister) ? (1 << (reg)) : 0; -} - -ParallelMoveResolver::ParallelMoveResolver() : moves_(32) {} - -void ParallelMoveResolver::Resolve(ParallelMoveInstr* parallel_move) { - ASSERT(moves_.is_empty()); - - // Build up a worklist of moves. - BuildInitialMoveList(parallel_move); - - const InstructionSource& move_source = InstructionSource( - TokenPosition::kParallelMove, parallel_move->inlining_id()); - for (intptr_t i = 0; i < moves_.length(); ++i) { - const MoveOperands& move = moves_[i]; - // Skip constants to perform them last. They don't block other moves - // and skipping such moves with register destinations keeps those - // registers free for the whole algorithm. - if (!move.IsEliminated() && !move.src().IsConstant()) { - PerformMove(move_source, i); - } - } - - // Perform the moves with constant sources. - for (const auto& move : moves_) { - if (!move.IsEliminated()) { - ASSERT(move.src().IsConstant()); - scheduled_ops_.Add({OpKind::kMove, move}); - } - } - moves_.Clear(); - - // Schedule is ready. Update parallel move itself. - parallel_move->set_move_schedule(MoveSchedule::From(scheduled_ops_)); - scheduled_ops_.Clear(); -} - -void ParallelMoveResolver::BuildInitialMoveList( - ParallelMoveInstr* parallel_move) { - // Perform a linear sweep of the moves to add them to the initial list of - // moves to perform, ignoring any move that is redundant (the source is - // the same as the destination, the destination is ignored and - // unallocated, or the move was already eliminated). - for (int i = 0; i < parallel_move->NumMoves(); i++) { - MoveOperands* move = parallel_move->MoveOperandsAt(i); - if (!move->IsRedundant()) moves_.Add(*move); - } -} - -void ParallelMoveResolver::PerformMove(const InstructionSource& source, - int index) { - // Each call to this function performs a move and deletes it from the move - // graph. We first recursively perform any move blocking this one. We - // mark a move as "pending" on entry to PerformMove in order to detect - // cycles in the move graph. We use operand swaps to resolve cycles, - // which means that a call to PerformMove could change any source operand - // in the move graph. - - ASSERT(!moves_[index].IsPending()); - ASSERT(!moves_[index].IsRedundant()); - - // Clear this move's destination to indicate a pending move. The actual - // destination is saved in a stack-allocated local. Recursion may allow - // multiple moves to be pending. - ASSERT(!moves_[index].src().IsInvalid()); - Location destination = moves_[index].MarkPending(); - - // Perform a depth-first traversal of the move graph to resolve - // dependencies. Any unperformed, unpending move with a source the same - // as this one's destination blocks this one so recursively perform all - // such moves. - for (int i = 0; i < moves_.length(); ++i) { - const MoveOperands& other_move = moves_[i]; - if (other_move.Blocks(destination) && !other_move.IsPending()) { - // Though PerformMove can change any source operand in the move graph, - // this call cannot create a blocking move via a swap (this loop does - // not miss any). Assume there is a non-blocking move with source A - // and this move is blocked on source B and there is a swap of A and - // B. Then A and B must be involved in the same cycle (or they would - // not be swapped). Since this move's destination is B and there is - // only a single incoming edge to an operand, this move must also be - // involved in the same cycle. In that case, the blocking move will - // be created but will be "pending" when we return from PerformMove. - PerformMove(source, i); - } - } - - // We are about to resolve this move and don't need it marked as - // pending, so restore its destination. - moves_[index].ClearPending(destination); - - // This move's source may have changed due to swaps to resolve cycles and - // so it may now be the last move in the cycle. If so remove it. - if (moves_[index].src().Equals(destination)) { - moves_[index].Eliminate(); - return; - } - - // The move may be blocked on a (at most one) pending move, in which case - // we have a cycle. Search for such a blocking move and perform a swap to - // resolve it. - for (auto& other_move : moves_) { - if (other_move.Blocks(destination)) { - ASSERT(other_move.IsPending()); - AddSwapToSchedule(index); - return; - } - } - - // This move is not blocked. - AddMoveToSchedule(index); -} - -void ParallelMoveResolver::AddMoveToSchedule(int index) { - auto& move = moves_[index]; - scheduled_ops_.Add({OpKind::kMove, move}); - move.Eliminate(); -} - -void ParallelMoveResolver::AddSwapToSchedule(int index) { - auto& move = moves_[index]; - const auto source = move.src(); - const auto destination = move.dest(); - - scheduled_ops_.Add({OpKind::kSwap, move}); - - // The swap of source and destination has executed a move from source to - // destination. - move.Eliminate(); - - // Any unperformed (including pending) move with a source of either - // this move's source or destination needs to have their source - // changed to reflect the state of affairs after the swap. - for (auto& other_move : moves_) { - if (other_move.Blocks(source)) { - other_move.set_src(destination); - } else if (other_move.Blocks(destination)) { - other_move.set_src(source); - } - } -} - -void ParallelMoveEmitter::EmitNativeCode() { - const auto& move_schedule = parallel_move_->move_schedule(); - for (intptr_t i = 0; i < move_schedule.length(); i++) { - current_move_ = i; - const auto& op = move_schedule[i]; - switch (op.kind) { - case ParallelMoveResolver::OpKind::kNop: - // |MoveSchedule::From| is expected to filter nops. - UNREACHABLE(); - break; - case ParallelMoveResolver::OpKind::kMove: - EmitMove(op.operands); - break; - case ParallelMoveResolver::OpKind::kSwap: - EmitSwap(op.operands); - break; - } - } -} - -void ParallelMoveEmitter::EmitMove(const MoveOperands& move) { - const Location src = move.src(); - const Location dst = move.dest(); - ParallelMoveEmitter::TemporaryAllocator temp(this, /*blocked=*/kNoRegister); - compiler_->EmitMove(dst, src, &temp); -#if defined(DEBUG) - // Allocating a scratch register here may cause stack spilling. Neither the - // source nor destination register should be SP-relative in that case. - for (const Location& loc : {dst, src}) { - ASSERT(!temp.DidAllocateTemporary() || !loc.HasStackIndex() || - loc.base_reg() != SPREG); - } -#endif -} - -bool ParallelMoveEmitter::IsScratchLocation(Location loc) { - const auto& move_schedule = parallel_move_->move_schedule(); - for (intptr_t i = current_move_; i < move_schedule.length(); i++) { - const auto& op = move_schedule[i]; - if (op.operands.src().Equals(loc) || - (op.kind == ParallelMoveResolver::OpKind::kSwap && - op.operands.dest().Equals(loc))) { - return false; - } - } - - for (intptr_t i = current_move_ + 1; i < move_schedule.length(); i++) { - const auto& op = move_schedule[i]; - if (op.kind == ParallelMoveResolver::OpKind::kMove && - op.operands.dest().Equals(loc)) { - return true; - } - } - - return false; -} - -intptr_t ParallelMoveEmitter::AllocateScratchRegister( - Location::Kind kind, - uword blocked_mask, - intptr_t first_free_register, - intptr_t last_free_register, - bool* spilled) { - COMPILE_ASSERT(static_cast(sizeof(blocked_mask)) * kBitsPerByte >= - kNumberOfFpuRegisters); - COMPILE_ASSERT(static_cast(sizeof(blocked_mask)) * kBitsPerByte >= - kNumberOfCpuRegisters); - intptr_t scratch = -1; - for (intptr_t reg = first_free_register; reg <= last_free_register; reg++) { - if ((((1 << reg) & blocked_mask) == 0) && - IsScratchLocation(Location::MachineRegisterLocation(kind, reg))) { - scratch = reg; - break; - } - } - - if (scratch == -1) { - *spilled = true; - for (intptr_t reg = first_free_register; reg <= last_free_register; reg++) { - if (((1 << reg) & blocked_mask) == 0) { - scratch = reg; - break; - } - } - } else { - *spilled = false; - } - - return scratch; -} - -ParallelMoveEmitter::ScratchFpuRegisterScope::ScratchFpuRegisterScope( - ParallelMoveEmitter* emitter, - FpuRegister blocked) - : emitter_(emitter), reg_(kNoFpuRegister), spilled_(false) { - COMPILE_ASSERT(FpuTMP != kNoFpuRegister); - uword blocked_mask = - ((blocked != kNoFpuRegister) ? 1 << blocked : 0) | 1 << FpuTMP; - reg_ = static_cast( - emitter_->AllocateScratchRegister(Location::kFpuRegister, blocked_mask, 0, - kNumberOfFpuRegisters - 1, &spilled_)); - - if (spilled_) { - emitter->SpillFpuScratch(reg_); - } -} - -ParallelMoveEmitter::ScratchFpuRegisterScope::~ScratchFpuRegisterScope() { - if (spilled_) { - emitter_->RestoreFpuScratch(reg_); - } -} - -ParallelMoveEmitter::TemporaryAllocator::TemporaryAllocator( - ParallelMoveEmitter* emitter, - Register blocked) - : emitter_(emitter), - blocked_(blocked), - reg_(kNoRegister), - spilled_(false) {} - -Register ParallelMoveEmitter::TemporaryAllocator::AllocateTemporary() { - ASSERT(reg_ == kNoRegister); - - uword blocked_mask = RegMaskBit(blocked_) | kReservedCpuRegisters; - if (emitter_->compiler_->intrinsic_mode()) { - // Block additional registers that must be preserved for intrinsics. - blocked_mask |= RegMaskBit(ARGS_DESC_REG); -#if !defined(TARGET_ARCH_IA32) - // Need to preserve CODE_REG to be able to store the PC marker - // and load the pool pointer. - blocked_mask |= RegMaskBit(CODE_REG); -#endif - } - reg_ = static_cast( - emitter_->AllocateScratchRegister(Location::kRegister, blocked_mask, 0, - kNumberOfCpuRegisters - 1, &spilled_)); - - if (spilled_) { - emitter_->SpillScratch(reg_); - } - - DEBUG_ONLY(allocated_ = true;) - return reg_; -} - -void ParallelMoveEmitter::TemporaryAllocator::ReleaseTemporary() { - if (spilled_) { - emitter_->RestoreScratch(reg_); - } - reg_ = kNoRegister; -} - -ParallelMoveEmitter::ScratchRegisterScope::ScratchRegisterScope( - ParallelMoveEmitter* emitter, - Register blocked) - : allocator_(emitter, blocked) { - reg_ = allocator_.AllocateTemporary(); -} - -ParallelMoveEmitter::ScratchRegisterScope::~ScratchRegisterScope() { - allocator_.ReleaseTemporary(); -} - -} // namespace dart diff --git a/runtime/vm/compiler/backend/parallel_move_resolver.h b/runtime/vm/compiler/backend/parallel_move_resolver.h deleted file mode 100644 index 98444cca79e..00000000000 --- a/runtime/vm/compiler/backend/parallel_move_resolver.h +++ /dev/null @@ -1,157 +0,0 @@ -// Copyright (c) 2023, the Dart project authors. Please see the AUTHORS file -// for details. All rights reserved. Use of this source code is governed by a -// BSD-style license that can be found in the LICENSE file. - -#ifndef RUNTIME_VM_COMPILER_BACKEND_PARALLEL_MOVE_RESOLVER_H_ -#define RUNTIME_VM_COMPILER_BACKEND_PARALLEL_MOVE_RESOLVER_H_ - -#if defined(DART_PRECOMPILED_RUNTIME) -#error "AOT runtime should not use compiler sources (including header files)" -#endif // defined(DART_PRECOMPILED_RUNTIME) - -#include "vm/allocation.h" -#include "vm/compiler/backend/flow_graph_compiler.h" -#include "vm/compiler/backend/locations.h" -#include "vm/constants.h" - -namespace dart { - -class MoveOperands; - -class ParallelMoveResolver : public ValueObject { - public: - ParallelMoveResolver(); - - // Schedule moves specified by the given parallel move and store the - // schedule on the parallel move itself. - void Resolve(ParallelMoveInstr* parallel_move); - - private: - // Build the initial list of moves. - void BuildInitialMoveList(ParallelMoveInstr* parallel_move); - - // Perform the move at the moves_ index in question (possibly requiring - // other moves to satisfy dependencies). - void PerformMove(const InstructionSource& source, int index); - - // Schedule a move and remove it from the move graph. - void AddMoveToSchedule(int index); - - // Schedule a swap of two operands. The move from - // source to destination is removed from the move graph. - void AddSwapToSchedule(int index); - - FlowGraphCompiler* compiler_; - - // List of moves not yet resolved. - GrowableArray moves_; - - enum class OpKind { - kNop, - kMove, - kSwap, - }; - - struct Op { - OpKind kind; - MoveOperands operands; - }; - - GrowableArray scheduled_ops_; - - friend class MoveSchedule; - friend class ParallelMoveEmitter; -}; - -class ParallelMoveEmitter : public ValueObject { - public: - ParallelMoveEmitter(FlowGraphCompiler* compiler, - ParallelMoveInstr* parallel_move) - : compiler_(compiler), parallel_move_(parallel_move) {} - - void EmitNativeCode(); - - private: - class ScratchFpuRegisterScope : public ValueObject { - public: - ScratchFpuRegisterScope(ParallelMoveEmitter* emitter, FpuRegister blocked); - ~ScratchFpuRegisterScope(); - - FpuRegister reg() const { return reg_; } - - private: - ParallelMoveEmitter* const emitter_; - FpuRegister reg_; - bool spilled_; - }; - - class TemporaryAllocator : public TemporaryRegisterAllocator { - public: - TemporaryAllocator(ParallelMoveEmitter* emitter, Register blocked); - - Register AllocateTemporary() override; - void ReleaseTemporary() override; - DEBUG_ONLY(bool DidAllocateTemporary() { return allocated_; }) - - virtual ~TemporaryAllocator() { ASSERT(reg_ == kNoRegister); } - - private: - ParallelMoveEmitter* const emitter_; - const Register blocked_; - Register reg_; - bool spilled_; - DEBUG_ONLY(bool allocated_ = false); - }; - - class ScratchRegisterScope : public ValueObject { - public: - ScratchRegisterScope(ParallelMoveEmitter* emitter, Register blocked); - ~ScratchRegisterScope(); - - Register reg() const { return reg_; } - - private: - TemporaryAllocator allocator_; - Register reg_; - }; - - bool IsScratchLocation(Location loc); - intptr_t AllocateScratchRegister(Location::Kind kind, - uword blocked_mask, - intptr_t first_free_register, - intptr_t last_free_register, - bool* spilled); - - void SpillScratch(Register reg); - void RestoreScratch(Register reg); - void SpillFpuScratch(FpuRegister reg); - void RestoreFpuScratch(FpuRegister reg); - - // Generate the code for a move from source to destination. - void EmitMove(const MoveOperands& move); - - void EmitSwap(const MoveOperands& swap); - - // Verify the move list before performing moves. - void Verify(); - - // Helpers for non-trivial source-destination combinations that cannot - // be handled by a single instruction. - void MoveMemoryToMemory(const compiler::Address& dst, - const compiler::Address& src); - void Exchange(Register reg, const compiler::Address& mem); - void Exchange(const compiler::Address& mem1, const compiler::Address& mem2); - void Exchange(Register reg, Register base_reg, intptr_t stack_offset); - void Exchange(Register base_reg1, - intptr_t stack_offset1, - Register base_reg2, - intptr_t stack_offset2); - - FlowGraphCompiler* const compiler_; - ParallelMoveInstr* parallel_move_; - intptr_t current_move_; -}; - -} // namespace dart - -#endif // RUNTIME_VM_COMPILER_BACKEND_PARALLEL_MOVE_RESOLVER_H_ diff --git a/runtime/vm/compiler/compiler_sources.gni b/runtime/vm/compiler/compiler_sources.gni index 9a051b6e4b8..beb7b8065c4 100644 --- a/runtime/vm/compiler/compiler_sources.gni +++ b/runtime/vm/compiler/compiler_sources.gni @@ -78,8 +78,6 @@ compiler_sources = [ "backend/locations_helpers_arm.h", "backend/loops.cc", "backend/loops.h", - "backend/parallel_move_resolver.cc", - "backend/parallel_move_resolver.h", "backend/range_analysis.cc", "backend/range_analysis.h", "backend/redundancy_elimination.cc", diff --git a/runtime/vm/compiler/graph_intrinsifier.cc b/runtime/vm/compiler/graph_intrinsifier.cc index 9e14d4644c9..1a9d6ec1467 100644 --- a/runtime/vm/compiler/graph_intrinsifier.cc +++ b/runtime/vm/compiler/graph_intrinsifier.cc @@ -52,16 +52,22 @@ static void EmitCodeFor(FlowGraphCompiler* compiler, FlowGraph* graph) { if (block->IsGraphEntry()) continue; // No code for graph entry needed. if (block->HasParallelMove()) { - block->parallel_move()->EmitNativeCode(compiler); + compiler->parallel_move_resolver()->EmitNativeCode( + block->parallel_move()); } for (ForwardInstructionIterator it(block); !it.Done(); it.Advance()) { Instruction* instr = it.Current(); if (FLAG_code_comments) compiler->EmitComment(instr); - // Calls are not supported in intrinsics code. - ASSERT(instr->IsParallelMove() || - (instr->locs() != nullptr && !instr->locs()->always_calls())); - instr->EmitNativeCode(compiler); + if (instr->IsParallelMove()) { + compiler->parallel_move_resolver()->EmitNativeCode( + instr->AsParallelMove()); + } else { + ASSERT(instr->locs() != NULL); + // Calls are not supported in intrinsics code. + ASSERT(!instr->locs()->always_calls()); + instr->EmitNativeCode(compiler); + } } } compiler->assembler()->Comment("Graph intrinsic end"); diff --git a/runtime/vm/globals.h b/runtime/vm/globals.h index 2e002464c33..80b830e821e 100644 --- a/runtime/vm/globals.h +++ b/runtime/vm/globals.h @@ -153,7 +153,7 @@ const intptr_t kOffsetOfPtr = 32; #define OPEN_ARRAY_START(type, align) \ do { \ const uword result = reinterpret_cast(this) + sizeof(*this); \ - ASSERT(Utils::IsAligned(result, alignof(align))); \ + ASSERT(Utils::IsAligned(result, sizeof(align))); \ return reinterpret_cast(result); \ } while (0)