Files
sdk/runtime/vm/compiler/backend/flow_graph_compiler_arm64.cc
T
Ryan Macnak d36adbacaf [vm] Remove the VM isolate.
The former contents of the VM isolate are now included into each isolate group. This makes each isolate group's heap independent, and in particular allows each heap to be allocated to a separate pointer cage (not done in this CL).

The duplicated stubs that allowed PC relative calls are removed, since the originals can now be the target of PC relative calls.

The bootstrapping needing to load an AppJIT or AppAOT snapshot is reduced to allocating the oddballs. The code is entirely dropped in the AOT runtime, but the JIT runtime still has it to allow for flags to affect the compilation of the stub code. Further refactoring might be able to remove this for the JIT runtime too, with only gen_snapshot knowing how to bootstrap.

Class serialization no longer distinguishes predefined classes.

The page containing null is marked as never-evacuate. null, false and true must not move because the compiler relies on their low bits having certain patterns for some optimizations. (Previously, the entire VM isolate heap never moved.)

Compaction is disabled for IA32. Due to register pressure, some stub calls must not use a scratch register and embed the address of Code.

The page containing the call-through-safepoint stub is frozen when running with --write-protect-code and the stub is created at runtime (instead of loaded from an AppJIT or AppAOT snapshot). This stub must remain executable even during a safepoint, as a foreign call might during return during a safepoint and only block after the stub directs it to the runtime.

The snapshot symbols are renamed to kDartSnapshotData and kDartSnapshotText. There is no need to distinguish the VM isolate's snapshot, and snaphots are per isolate group not per isolate. Aliases with the old names are added to ease migration.

Some global flags that were automatically set based on the VM isolate's snapshot are now isolate group flags and automatically set by the isolate group's snapshot.

TEST=ci
Change-Id: Iee82016057d609112e9b021d178fc3d4d18b5044
Reviewed-on: https://dart-review.googlesource.com/c/sdk/+/500621
Reviewed-by: Alexander Markov <alexmarkov@google.com>
Reviewed-by: Tess Strickland <sstrickl@google.com>
SLSA-Policy-Verified: SLSA Policy Verification Service <devtools-gerritcodereview-exitgate@google.com>
Commit-Queue: Ryan Macnak <rmacnak@google.com>
2026-05-18 11:35:03 -07:00

1201 lines
45 KiB
C++

// Copyright (c) 2014, the Dart project authors. Please see the AUTHORS file
// for details. All rights reserved. Use of this source code is governed by a
// BSD-style license that can be found in the LICENSE file.
#include "vm/globals.h" // Needed here to get TARGET_ARCH_ARM64.
#if defined(TARGET_ARCH_ARM64)
#include "vm/compiler/backend/flow_graph_compiler.h"
#include "vm/compiler/api/type_check_mode.h"
#include "vm/compiler/backend/il_printer.h"
#include "vm/compiler/backend/locations.h"
#include "vm/compiler/backend/parallel_move_resolver.h"
#include "vm/compiler/jit/compiler.h"
#include "vm/cpu.h"
#include "vm/dart_entry.h"
#include "vm/deopt_instructions.h"
#include "vm/dispatch_table.h"
#include "vm/instructions.h"
#include "vm/object_store.h"
#include "vm/parser.h"
#include "vm/stack_frame.h"
#include "vm/stub_code.h"
#include "vm/symbols.h"
namespace dart {
DEFINE_FLAG(bool, trap_on_deoptimization, false, "Trap on deoptimization.");
DECLARE_FLAG(bool, enable_simd_inline);
void FlowGraphCompiler::ArchSpecificInitialization() {
if (FLAG_precompiled_mode) {
const auto& stub = StubCode::WriteBarrierWrappers();
if (CanPcRelativeCall(stub)) {
assembler_->generate_invoke_write_barrier_wrapper_ = [&](Register reg) {
const intptr_t offset_into_target =
Thread::WriteBarrierWrappersOffsetForRegister(reg);
assembler_->GenerateUnRelocatedPcRelativeCall(offset_into_target);
AddPcRelativeCallStubTarget(stub);
};
}
const auto& array_stub = StubCode::ArrayWriteBarrier();
if (CanPcRelativeCall(stub)) {
assembler_->generate_invoke_array_write_barrier_ = [&]() {
assembler_->GenerateUnRelocatedPcRelativeCall();
AddPcRelativeCallStubTarget(array_stub);
};
}
}
}
FlowGraphCompiler::~FlowGraphCompiler() {
// BlockInfos are zone-allocated, so their destructors are not called.
// Verify the labels explicitly here.
for (int i = 0; i < block_info_.length(); ++i) {
ASSERT(!block_info_[i]->jump_label()->IsLinked());
}
}
bool FlowGraphCompiler::SupportsUnboxedSimd128() {
return FLAG_enable_simd_inline;
}
bool FlowGraphCompiler::CanConvertInt64ToDouble() {
return true;
}
void FlowGraphCompiler::EnterIntrinsicMode() {
ASSERT(!intrinsic_mode());
intrinsic_mode_ = true;
ASSERT(!assembler()->constant_pool_allowed());
}
void FlowGraphCompiler::ExitIntrinsicMode() {
ASSERT(intrinsic_mode());
intrinsic_mode_ = false;
}
TypedDataPtr CompilerDeoptInfo::CreateDeoptInfo(FlowGraphCompiler* compiler,
DeoptInfoBuilder* builder,
const Array& deopt_table) {
if (deopt_env_ == nullptr) {
++builder->current_info_number_;
return TypedData::null();
}
AllocateOutgoingArguments(deopt_env_);
intptr_t slot_ix = 0;
Environment* current = deopt_env_;
// Emit all kMaterializeObject instructions describing objects to be
// materialized on the deoptimization as a prefix to the deoptimization info.
EmitMaterializations(deopt_env_, builder);
// The real frame starts here.
builder->MarkFrameStart();
Zone* zone = compiler->zone();
builder->AddPp(current->function(), slot_ix++);
builder->AddPcMarker(Function::ZoneHandle(zone), slot_ix++);
builder->AddCallerFp(slot_ix++);
builder->AddReturnAddress(current->function(), deopt_id(), slot_ix++);
// Emit all values that are needed for materialization as a part of the
// expression stack for the bottom-most frame. This guarantees that GC
// will be able to find them during materialization.
slot_ix = builder->EmitMaterializationArguments(slot_ix);
// For the innermost environment, set outgoing arguments and the locals.
for (intptr_t i = current->Length() - 1;
i >= current->fixed_parameter_count(); i--) {
builder->AddCopy(current->ValueAt(i), current->LocationAt(i), slot_ix++);
}
Environment* previous = current;
current = current->outer();
while (current != nullptr) {
builder->AddPp(current->function(), slot_ix++);
builder->AddPcMarker(previous->function(), slot_ix++);
builder->AddCallerFp(slot_ix++);
// For any outer environment the deopt id is that of the call instruction
// which is recorded in the outer environment.
builder->AddReturnAddress(current->function(),
DeoptId::ToDeoptAfter(current->GetDeoptId()),
slot_ix++);
// The values of outgoing arguments can be changed from the inlined call so
// we must read them from the previous environment.
for (intptr_t i = previous->fixed_parameter_count() - 1; i >= 0; i--) {
builder->AddCopy(previous->ValueAt(i), previous->LocationAt(i),
slot_ix++);
}
// Set the locals, note that outgoing arguments are not in the environment.
for (intptr_t i = current->Length() - 1;
i >= current->fixed_parameter_count(); i--) {
builder->AddCopy(current->ValueAt(i), current->LocationAt(i), slot_ix++);
}
// Iterate on the outer environment.
previous = current;
current = current->outer();
}
// The previous pointer is now the outermost environment.
ASSERT(previous != nullptr);
// Add slots for the outermost environment.
builder->AddCallerPp(slot_ix++);
builder->AddPcMarker(previous->function(), slot_ix++);
builder->AddCallerFp(slot_ix++);
builder->AddCallerPc(slot_ix++);
// For the outermost environment, set the incoming arguments.
for (intptr_t i = previous->fixed_parameter_count() - 1; i >= 0; i--) {
builder->AddCopy(previous->ValueAt(i), previous->LocationAt(i), slot_ix++);
}
return builder->CreateDeoptInfo(deopt_table);
}
void CompilerDeoptInfoWithStub::GenerateCode(FlowGraphCompiler* compiler,
intptr_t stub_ix) {
// Calls do not need stubs, they share a deoptimization trampoline.
ASSERT(reason() != ICData::kDeoptAtCall);
compiler::Assembler* assembler = compiler->assembler();
#define __ assembler->
__ Comment("%s", Name());
__ Bind(entry_label());
if (FLAG_trap_on_deoptimization) {
__ brk(0);
}
ASSERT(deopt_env() != nullptr);
__ Call(compiler::Address(THR, Thread::deoptimize_entry_offset()));
set_pc_offset(assembler->CodeSize());
#undef __
}
#define __ assembler->
// Static methods of FlowGraphCompiler that take an assembler.
void FlowGraphCompiler::GenerateIndirectTTSCall(compiler::Assembler* assembler,
Register reg_to_call,
intptr_t sub_type_cache_index) {
__ LoadField(
TTSInternalRegs::kScratchReg,
compiler::FieldAddress(
reg_to_call,
compiler::target::AbstractType::type_test_stub_entry_point_offset()));
__ LoadWordFromPoolIndex(TypeTestABI::kSubtypeTestCacheReg,
sub_type_cache_index);
__ blr(TTSInternalRegs::kScratchReg);
}
#undef __
#define __ assembler()->
// Instance methods of FlowGraphCompiler.
// Fall through if bool_register contains null.
void FlowGraphCompiler::GenerateBoolToJump(Register bool_register,
compiler::Label* is_true,
compiler::Label* is_false) {
compiler::Label fall_through;
__ CompareObject(bool_register, Object::null_object());
__ b(&fall_through, EQ);
BranchLabels labels = {is_true, is_false, &fall_through};
Condition true_condition =
EmitBoolTest(bool_register, labels, /*invert=*/false);
ASSERT(true_condition == kInvalidCondition);
__ Bind(&fall_through);
}
void FlowGraphCompiler::EmitFrameEntry() {
const Function& function = parsed_function().function();
if (CanOptimizeFunction() && function.IsOptimizable() &&
(!is_optimizing() || may_reoptimize())) {
__ Comment("Invocation Count Check");
const Register function_reg = R6;
__ ldr(function_reg,
compiler::FieldAddress(CODE_REG, Code::owner_offset()));
__ LoadFieldFromOffset(R7, function_reg, Function::usage_counter_offset(),
compiler::kFourBytes);
// Reoptimization of an optimized function is triggered by counting in
// IC stubs, but not at the entry of the function.
if (!is_optimizing()) {
__ add(R7, R7, compiler::Operand(1));
__ StoreFieldToOffset(R7, function_reg, Function::usage_counter_offset(),
compiler::kFourBytes);
}
__ CompareImmediate(R7, GetOptimizationThreshold());
ASSERT(function_reg == R6);
compiler::Label dont_optimize;
__ b(&dont_optimize, LT);
__ ldr(TMP, compiler::Address(THR, Thread::optimize_entry_offset()));
__ br(TMP);
__ Bind(&dont_optimize);
}
if (flow_graph().graph_entry()->NeedsFrame()) {
__ Comment("Enter frame");
if (flow_graph().IsCompiledForOsr()) {
const intptr_t extra_slots = ExtraStackSlotsOnOsrEntry();
ASSERT(extra_slots >= 0);
__ EnterOsrFrame(extra_slots * kWordSize);
} else {
ASSERT(StackSize() >= 0);
__ EnterDartFrame(StackSize() * kWordSize);
}
} else if (FLAG_precompiled_mode) {
assembler()->set_constant_pool_allowed(true);
}
if (FLAG_target_thread_sanitizer && !is_optimizing()) {
bool uses_args_desc = parsed_function().has_arg_desc_var();
if (uses_args_desc) {
__ MoveRegister(CALLEE_SAVED_TEMP, ARGS_DESC_REG);
}
__ TsanFuncEntry(/*preserve_registers=*/false);
if (uses_args_desc) {
__ MoveRegister(ARGS_DESC_REG, CALLEE_SAVED_TEMP);
}
}
}
const InstructionSource& PrologueSource() {
static InstructionSource prologue_source(TokenPosition::kDartCodePrologue,
/*inlining_id=*/0);
return prologue_source;
}
void FlowGraphCompiler::EmitPrologue() {
BeginCodeSourceRange(PrologueSource());
EmitFrameEntry();
ASSERT(assembler()->constant_pool_allowed());
// In unoptimized code, initialize (non-argument) stack allocated slots.
if (!is_optimizing()) {
const int num_locals = parsed_function().num_stack_locals();
intptr_t args_desc_slot = -1;
if (parsed_function().has_arg_desc_var()) {
args_desc_slot = compiler::target::frame_layout.FrameSlotForVariable(
parsed_function().arg_desc_var());
}
__ Comment("Initialize spill slots");
for (intptr_t i = 0; i < num_locals; ++i) {
const intptr_t slot_index =
compiler::target::frame_layout.FrameSlotForVariableIndex(-i);
Register value_reg =
slot_index == args_desc_slot ? ARGS_DESC_REG : NULL_REG;
__ StoreToOffset(value_reg, FP, slot_index * kWordSize);
}
} else if (parsed_function().suspend_state_var() != nullptr &&
!flow_graph().IsCompiledForOsr()) {
// Initialize synthetic :suspend_state variable early
// as it may be accessed by GC and exception handling before
// InitSuspendableFunction stub is called.
const intptr_t slot_index =
compiler::target::frame_layout.FrameSlotForVariable(
parsed_function().suspend_state_var());
__ StoreToOffset(NULL_REG, FP, slot_index * kWordSize);
}
EndCodeSourceRange(PrologueSource());
}
void FlowGraphCompiler::EmitCallToStub(
const Code& stub,
ObjectPool::SnapshotBehavior snapshot_behavior) {
ASSERT(!stub.IsNull());
if (CanPcRelativeCall(stub)) {
__ GenerateUnRelocatedPcRelativeCall();
AddPcRelativeCallStubTarget(stub);
} else {
__ BranchLink(stub, compiler::ObjectPoolBuilderEntry::kNotPatchable,
CodeEntryKind::kNormal, snapshot_behavior);
AddStubCallTarget(stub);
}
}
void FlowGraphCompiler::EmitJumpToStub(const Code& stub) {
ASSERT(!stub.IsNull());
if (CanPcRelativeCall(stub)) {
__ GenerateUnRelocatedPcRelativeTailCall();
AddPcRelativeTailCallStubTarget(stub);
} else {
__ LoadObject(CODE_REG, stub);
__ ldr(TMP, compiler::FieldAddress(
CODE_REG, compiler::target::Code::entry_point_offset()));
__ br(TMP);
AddStubCallTarget(stub);
}
}
void FlowGraphCompiler::EmitTailCallToStub(const Code& stub) {
ASSERT(!stub.IsNull());
if (CanPcRelativeCall(stub)) {
if (flow_graph().graph_entry()->NeedsFrame()) {
if (FLAG_target_thread_sanitizer && !is_optimizing()) {
__ TsanFuncExit();
}
__ LeaveDartFrame();
}
__ GenerateUnRelocatedPcRelativeTailCall();
AddPcRelativeTailCallStubTarget(stub);
#if defined(DEBUG)
__ Breakpoint();
#endif
} else {
__ LoadObject(CODE_REG, stub);
if (flow_graph().graph_entry()->NeedsFrame()) {
if (FLAG_target_thread_sanitizer && !is_optimizing()) {
__ TsanFuncExit();
}
__ LeaveDartFrame();
}
__ ldr(TMP, compiler::FieldAddress(
CODE_REG, compiler::target::Code::entry_point_offset()));
__ br(TMP);
AddStubCallTarget(stub);
}
}
void FlowGraphCompiler::GeneratePatchableCall(
const InstructionSource& source,
const Code& stub,
UntaggedPcDescriptors::Kind kind,
LocationSummary* locs,
ObjectPool::SnapshotBehavior snapshot_behavior) {
__ BranchLinkPatchable(stub, CodeEntryKind::kNormal, snapshot_behavior);
EmitCallsiteMetadata(source, DeoptId::kNone, kind, locs,
pending_deoptimization_env_);
}
void FlowGraphCompiler::GenerateDartCall(intptr_t deopt_id,
const InstructionSource& source,
const Code& stub,
UntaggedPcDescriptors::Kind kind,
LocationSummary* locs,
Code::EntryKind entry_kind) {
ASSERT(CanCallDart());
__ BranchLinkPatchable(stub, entry_kind);
EmitCallsiteMetadata(source, deopt_id, kind, locs,
pending_deoptimization_env_);
}
void FlowGraphCompiler::GenerateStaticDartCall(intptr_t deopt_id,
const InstructionSource& source,
UntaggedPcDescriptors::Kind kind,
LocationSummary* locs,
const Function& target,
Code::EntryKind entry_kind) {
ASSERT(CanCallDart());
if (CanPcRelativeCall(target)) {
__ GenerateUnRelocatedPcRelativeCall();
AddPcRelativeCallTarget(target, entry_kind);
EmitCallsiteMetadata(source, deopt_id, kind, locs,
pending_deoptimization_env_);
} else {
// Call sites to the same target can share object pool entries. These
// call sites are never patched for breakpoints: the function is deoptimized
// and the unoptimized code with IC calls for static calls is patched
// instead.
ASSERT(is_optimizing());
const auto& stub = StubCode::CallStaticFunction();
__ BranchLinkWithEquivalence(stub, target, entry_kind);
EmitCallsiteMetadata(source, deopt_id, kind, locs,
pending_deoptimization_env_);
AddStaticCallTarget(target, entry_kind);
}
}
void FlowGraphCompiler::EmitEdgeCounter(intptr_t edge_id) {
// We do not check for overflow when incrementing the edge counter. The
// function should normally be optimized long before the counter can
// overflow; and though we do not reset the counters when we optimize or
// deoptimize, there is a bound on the number of
// optimization/deoptimization cycles we will attempt.
ASSERT(!edge_counters_array_.IsNull());
ASSERT(assembler_->constant_pool_allowed());
__ Comment("Edge counter");
__ LoadObject(R0, edge_counters_array_);
__ LoadCompressedSmiFieldFromOffset(TMP, R0, Array::element_offset(edge_id));
__ add(TMP, TMP, compiler::Operand(Smi::RawValue(1)), compiler::kObjectBytes);
__ StoreFieldToOffset(TMP, R0, Array::element_offset(edge_id),
compiler::kObjectBytes);
}
void FlowGraphCompiler::EmitOptimizedInstanceCall(
const Code& stub,
const ICData& ic_data,
intptr_t deopt_id,
const InstructionSource& source,
LocationSummary* locs,
Code::EntryKind entry_kind) {
ASSERT(CanCallDart());
ASSERT(Array::Handle(zone(), ic_data.arguments_descriptor()).Length() > 0);
// Each ICData propagated from unoptimized to optimized code contains the
// function that corresponds to the Dart function of that IC call. Due
// to inlining in optimized code, that function may not correspond to the
// top-level function (parsed_function().function()) which could be
// reoptimized and which counter needs to be incremented.
// Pass the function explicitly, it is used in IC stub.
__ LoadObject(R6, parsed_function().function());
__ LoadFromOffset(R0, SP, (ic_data.SizeWithoutTypeArgs() - 1) * kWordSize);
__ LoadUniqueObject(IC_DATA_REG, ic_data);
GenerateDartCall(deopt_id, source, stub, UntaggedPcDescriptors::kIcCall, locs,
entry_kind);
EmitDropArguments(ic_data.SizeWithTypeArgs());
}
void FlowGraphCompiler::EmitInstanceCallJIT(const Code& stub,
const ICData& ic_data,
intptr_t deopt_id,
const InstructionSource& source,
LocationSummary* locs,
Code::EntryKind entry_kind) {
ASSERT(CanCallDart());
ASSERT(entry_kind == Code::EntryKind::kNormal ||
entry_kind == Code::EntryKind::kUnchecked);
ASSERT(Array::Handle(zone(), ic_data.arguments_descriptor()).Length() > 0);
__ LoadFromOffset(R0, SP, (ic_data.SizeWithoutTypeArgs() - 1) * kWordSize);
compiler::ObjectPoolBuilder& op = __ object_pool_builder();
const intptr_t stub_index =
op.AddObject(stub, ObjectPool::Patchability::kPatchable);
const intptr_t ic_data_index =
op.AddObject(ic_data, ObjectPool::Patchability::kPatchable);
ASSERT((stub_index + 1) == ic_data_index);
__ LoadDoubleWordFromPoolIndex(CODE_REG, IC_DATA_REG, stub_index);
const intptr_t entry_point_offset =
entry_kind == Code::EntryKind::kNormal
? Code::entry_point_offset(Code::EntryKind::kMonomorphic)
: Code::entry_point_offset(Code::EntryKind::kMonomorphicUnchecked);
__ Call(compiler::FieldAddress(CODE_REG, entry_point_offset));
EmitCallsiteMetadata(source, deopt_id, UntaggedPcDescriptors::kIcCall, locs,
pending_deoptimization_env_);
EmitDropArguments(ic_data.SizeWithTypeArgs());
}
void FlowGraphCompiler::EmitMegamorphicInstanceCall(
const String& name,
const Array& arguments_descriptor,
intptr_t deopt_id,
const InstructionSource& source,
LocationSummary* locs) {
ASSERT(CanCallDart());
ASSERT(!arguments_descriptor.IsNull() && (arguments_descriptor.Length() > 0));
ASSERT(!FLAG_precompiled_mode);
const ArgumentsDescriptor args_desc(arguments_descriptor);
const MegamorphicCache& cache = MegamorphicCache::ZoneHandle(
zone(),
MegamorphicCacheTable::Lookup(thread(), name, arguments_descriptor));
__ Comment("MegamorphicCall");
// Load receiver into R0.
__ LoadFromOffset(R0, SP, (args_desc.Count() - 1) * kWordSize);
// Use same code pattern as instance call so it can be parsed by code patcher.
compiler::ObjectPoolBuilder& op = __ object_pool_builder();
const intptr_t stub_index = op.AddObject(
StubCode::MegamorphicCall(), ObjectPool::Patchability::kPatchable);
const intptr_t data_index =
op.AddObject(cache, ObjectPool::Patchability::kPatchable);
ASSERT((stub_index + 1) == data_index);
__ LoadDoubleWordFromPoolIndex(CODE_REG, IC_DATA_REG, stub_index);
CLOBBERS_LR(__ ldr(LR, compiler::FieldAddress(
CODE_REG, Code::entry_point_offset(
Code::EntryKind::kMonomorphic))));
CLOBBERS_LR(__ blr(LR));
RecordSafepoint(locs);
AddCurrentDescriptor(UntaggedPcDescriptors::kOther, DeoptId::kNone, source);
const intptr_t deopt_id_after = DeoptId::ToDeoptAfter(deopt_id);
if (is_optimizing()) {
AddDeoptIndexAtCall(deopt_id_after, pending_deoptimization_env_);
} else {
// Add deoptimization continuation point after the call and before the
// arguments are removed.
AddCurrentDescriptor(UntaggedPcDescriptors::kDeopt, deopt_id_after, source);
}
RecordCatchEntryMoves(pending_deoptimization_env_);
EmitDropArguments(args_desc.SizeWithTypeArgs());
}
void FlowGraphCompiler::EmitInstanceCallAOT(const ICData& ic_data,
intptr_t deopt_id,
const InstructionSource& source,
LocationSummary* locs,
Code::EntryKind entry_kind,
bool receiver_can_be_smi) {
ASSERT(CanCallDart());
ASSERT(ic_data.NumArgsTested() == 1);
const Code& initial_stub = StubCode::SwitchableCallMiss();
const char* switchable_call_mode = "smiable";
if (!receiver_can_be_smi) {
switchable_call_mode = "non-smi";
ic_data.set_receiver_cannot_be_smi(true);
}
const UnlinkedCall& data =
UnlinkedCall::ZoneHandle(zone(), ic_data.AsUnlinkedCall());
compiler::ObjectPoolBuilder& op = __ object_pool_builder();
__ Comment("InstanceCallAOT (%s)", switchable_call_mode);
// Clear argument descriptor to keep gc happy when it gets pushed on to
// the stack.
__ LoadImmediate(R4, 0);
__ LoadFromOffset(R0, SP, (ic_data.SizeWithoutTypeArgs() - 1) * kWordSize);
const auto snapshot_behavior =
FLAG_precompiled_mode ? compiler::ObjectPoolBuilderEntry::
kResetToSwitchableCallMissEntryPoint
: compiler::ObjectPoolBuilderEntry::kSnapshotable;
const intptr_t stub_index = op.AddObject(
initial_stub, ObjectPool::Patchability::kPatchable, snapshot_behavior);
const intptr_t data_index =
op.AddObject(data, ObjectPool::Patchability::kPatchable);
ASSERT((stub_index + 1) == data_index);
// The AOT runtime will replace the slot in the object pool with the
// entrypoint address - see app_snapshot.cc.
CLOBBERS_LR(__ LoadDoubleWordFromPoolIndex(LR, R5, stub_index));
CLOBBERS_LR(__ blr(LR));
EmitCallsiteMetadata(source, DeoptId::kNone, UntaggedPcDescriptors::kOther,
locs, pending_deoptimization_env_);
EmitDropArguments(ic_data.SizeWithTypeArgs());
}
void FlowGraphCompiler::EmitUnoptimizedStaticCall(
intptr_t size_with_type_args,
intptr_t deopt_id,
const InstructionSource& source,
LocationSummary* locs,
const ICData& ic_data,
Code::EntryKind entry_kind) {
ASSERT(CanCallDart());
const Code& stub =
StubCode::UnoptimizedStaticCallEntry(ic_data.NumArgsTested());
__ LoadObject(R5, ic_data);
GenerateDartCall(deopt_id, source, stub,
UntaggedPcDescriptors::kUnoptStaticCall, locs, entry_kind);
EmitDropArguments(size_with_type_args);
}
void FlowGraphCompiler::EmitOptimizedStaticCall(
const Function& function,
const Array& arguments_descriptor,
intptr_t size_with_type_args,
intptr_t deopt_id,
const InstructionSource& source,
LocationSummary* locs,
Code::EntryKind entry_kind) {
ASSERT(CanCallDart());
ASSERT(!function.IsClosureFunction());
if (function.PrologueNeedsArgumentsDescriptor()) {
__ LoadObject(ARGS_DESC_REG, arguments_descriptor);
} else {
if (!FLAG_precompiled_mode) {
__ LoadImmediate(ARGS_DESC_REG, 0); // GC safe smi zero because of stub.
}
}
// Do not use the code from the function, but let the code be patched so that
// we can record the outgoing edges to other code.
GenerateStaticDartCall(deopt_id, source, UntaggedPcDescriptors::kOther, locs,
function, entry_kind);
EmitDropArguments(size_with_type_args);
}
void FlowGraphCompiler::EmitDispatchTableCall(
int32_t selector_offset,
const Array& arguments_descriptor) {
const auto cid_reg = DispatchTableNullErrorABI::kClassIdReg;
ASSERT(CanCallDart());
ASSERT(cid_reg != ARGS_DESC_REG);
if (!arguments_descriptor.IsNull()) {
__ LoadObject(ARGS_DESC_REG, arguments_descriptor);
}
const intptr_t offset = selector_offset - DispatchTable::kOriginElement;
CLOBBERS_LR({
// Would like cid_reg to be available on entry to the target function
// for checking purposes.
ASSERT(cid_reg != LR);
__ AddImmediate(LR, cid_reg, offset);
__ Call(compiler::Address(DISPATCH_TABLE_REG, LR, UXTX,
compiler::Address::Scaled));
});
}
Condition FlowGraphCompiler::EmitEqualityRegConstCompare(
Register reg,
const Object& obj,
bool needs_number_check,
const InstructionSource& source,
intptr_t deopt_id) {
if (needs_number_check) {
ASSERT(!obj.IsMint() && !obj.IsDouble());
__ LoadObject(TMP, obj);
__ PushPair(TMP, reg);
if (is_optimizing()) {
// No breakpoints in optimized code.
__ BranchLink(StubCode::OptimizedIdenticalWithNumberCheck());
AddCurrentDescriptor(UntaggedPcDescriptors::kOther, deopt_id, source);
} else {
// Patchable to support breakpoints.
__ BranchLinkPatchable(StubCode::UnoptimizedIdenticalWithNumberCheck());
AddCurrentDescriptor(UntaggedPcDescriptors::kRuntimeCall, deopt_id,
source);
}
// Stub returns result in flags (result of a cmp, we need Z computed).
// Discard constant.
// Restore 'reg'.
__ PopPair(ZR, reg);
} else {
__ CompareObject(reg, obj);
}
return EQ;
}
Condition FlowGraphCompiler::EmitEqualityRegRegCompare(
Register left,
Register right,
bool needs_number_check,
const InstructionSource& source,
intptr_t deopt_id) {
if (needs_number_check) {
__ PushPair(right, left);
if (is_optimizing()) {
__ BranchLink(StubCode::OptimizedIdenticalWithNumberCheck());
} else {
__ BranchLinkPatchable(StubCode::UnoptimizedIdenticalWithNumberCheck());
}
AddCurrentDescriptor(UntaggedPcDescriptors::kRuntimeCall, deopt_id, source);
// Stub returns result in flags (result of a cmp, we need Z computed).
__ PopPair(right, left);
} else {
__ CompareObjectRegisters(left, right);
}
return EQ;
}
Condition FlowGraphCompiler::EmitBoolTest(Register value,
BranchLabels labels,
bool invert) {
__ Comment("BoolTest");
if (labels.true_label == nullptr || labels.false_label == nullptr) {
__ tsti(value, compiler::Immediate(
compiler::target::ObjectAlignment::kBoolValueMask));
return invert ? NE : EQ;
}
const intptr_t bool_bit =
compiler::target::ObjectAlignment::kBoolValueBitPosition;
if (labels.fall_through == labels.false_label) {
if (invert) {
__ tbnz(labels.true_label, value, bool_bit);
} else {
__ tbz(labels.true_label, value, bool_bit);
}
} else {
if (invert) {
__ tbz(labels.false_label, value, bool_bit);
} else {
__ tbnz(labels.false_label, value, bool_bit);
}
if (labels.fall_through != labels.true_label) {
__ b(labels.true_label);
}
}
return kInvalidCondition;
}
// This function must be in sync with FlowGraphCompiler::RecordSafepoint and
// FlowGraphCompiler::SlowPathEnvironmentFor.
void FlowGraphCompiler::SaveLiveRegisters(LocationSummary* locs) {
#if defined(DEBUG)
locs->CheckWritableInputs();
ClobberDeadTempRegisters(locs);
#endif
// TODO(vegorov): consider saving only caller save (volatile) registers.
__ PushRegisters(*locs->live_registers());
}
void FlowGraphCompiler::RestoreLiveRegisters(LocationSummary* locs) {
__ PopRegisters(*locs->live_registers());
}
#if defined(DEBUG)
void FlowGraphCompiler::ClobberDeadTempRegisters(LocationSummary* locs) {
// Clobber temporaries that have not been manually preserved.
for (intptr_t i = 0; i < locs->temp_count(); ++i) {
Location tmp = locs->temp(i);
// TODO(zerny): clobber non-live temporary FPU registers.
if (tmp.IsRegister() &&
!locs->live_registers()->ContainsRegister(tmp.reg())) {
__ movz(tmp.reg(), compiler::Immediate(0xf7), 0);
}
}
}
#endif
Register FlowGraphCompiler::EmitTestCidRegister() {
return R2;
}
void FlowGraphCompiler::EmitTestAndCallLoadReceiver(
intptr_t count_without_type_args,
const Array& arguments_descriptor) {
__ Comment("EmitTestAndCall");
// Load receiver into R0.
__ LoadFromOffset(R0, SP, (count_without_type_args - 1) * kWordSize);
__ LoadObject(ARGS_DESC_REG, arguments_descriptor);
}
void FlowGraphCompiler::EmitTestAndCallSmiBranch(compiler::Label* label,
bool if_smi) {
if (if_smi) {
__ BranchIfSmi(R0, label);
} else {
__ BranchIfNotSmi(R0, label);
}
}
void FlowGraphCompiler::EmitTestAndCallLoadCid(Register class_id_reg) {
ASSERT(class_id_reg != R0);
__ LoadClassId(class_id_reg, R0);
}
void FlowGraphCompiler::EmitMove(Location destination,
Location source,
TemporaryRegisterAllocator* allocator) {
if (destination.Equals(source)) return;
if (source.IsRegister()) {
if (destination.IsRegister()) {
__ mov(destination.reg(), source.reg());
} else {
ASSERT(destination.IsStackSlot());
const intptr_t dest_offset = destination.ToStackSlotOffset();
__ StoreToOffset(source.reg(), destination.base_reg(), dest_offset);
}
} else if (source.IsStackSlot()) {
if (destination.IsRegister()) {
const intptr_t source_offset = source.ToStackSlotOffset();
__ LoadFromOffset(destination.reg(), source.base_reg(), source_offset);
} else if (destination.IsFpuRegister()) {
const intptr_t src_offset = source.ToStackSlotOffset();
VRegister dst = destination.fpu_reg();
__ LoadDFromOffset(dst, source.base_reg(), src_offset);
} else {
ASSERT(destination.IsStackSlot());
const intptr_t source_offset = source.ToStackSlotOffset();
const intptr_t dest_offset = destination.ToStackSlotOffset();
Register tmp = allocator->AllocateTemporary();
__ LoadFromOffset(tmp, source.base_reg(), source_offset);
__ StoreToOffset(tmp, destination.base_reg(), dest_offset);
allocator->ReleaseTemporary();
}
} else if (source.IsFpuRegister()) {
if (destination.IsFpuRegister()) {
__ vmov(destination.fpu_reg(), source.fpu_reg());
} else {
if (destination.IsStackSlot() /*32-bit float*/ ||
destination.IsDoubleStackSlot()) {
const intptr_t dest_offset = destination.ToStackSlotOffset();
VRegister src = source.fpu_reg();
__ StoreDToOffset(src, destination.base_reg(), dest_offset);
} else {
ASSERT(destination.IsQuadStackSlot());
const intptr_t dest_offset = destination.ToStackSlotOffset();
__ StoreQToOffset(source.fpu_reg(), destination.base_reg(),
dest_offset);
}
}
} else if (source.IsDoubleStackSlot()) {
if (destination.IsFpuRegister()) {
const intptr_t source_offset = source.ToStackSlotOffset();
const VRegister dst = destination.fpu_reg();
__ LoadDFromOffset(dst, source.base_reg(), source_offset);
} else {
ASSERT(destination.IsDoubleStackSlot() ||
destination.IsStackSlot() /*32-bit float*/);
const intptr_t source_offset = source.ToStackSlotOffset();
const intptr_t dest_offset = destination.ToStackSlotOffset();
__ LoadDFromOffset(VTMP, source.base_reg(), source_offset);
__ StoreDToOffset(VTMP, destination.base_reg(), dest_offset);
}
} else if (source.IsQuadStackSlot()) {
if (destination.IsFpuRegister()) {
const intptr_t source_offset = source.ToStackSlotOffset();
__ LoadQFromOffset(destination.fpu_reg(), source.base_reg(),
source_offset);
} else {
ASSERT(destination.IsQuadStackSlot());
const intptr_t source_offset = source.ToStackSlotOffset();
const intptr_t dest_offset = destination.ToStackSlotOffset();
__ LoadQFromOffset(VTMP, source.base_reg(), source_offset);
__ StoreQToOffset(VTMP, destination.base_reg(), dest_offset);
}
} else {
ASSERT(source.IsConstant());
if (destination.IsStackSlot()) {
Register tmp = allocator->AllocateTemporary();
source.constant_instruction()->EmitMoveToLocation(this, destination, tmp);
allocator->ReleaseTemporary();
} else {
source.constant_instruction()->EmitMoveToLocation(this, destination);
}
}
}
static compiler::OperandSize BytesToOperandSize(intptr_t bytes) {
switch (bytes) {
case 8:
return compiler::OperandSize::kEightBytes;
case 4:
return compiler::OperandSize::kFourBytes;
case 2:
return compiler::OperandSize::kTwoBytes;
case 1:
return compiler::OperandSize::kByte;
default:
UNIMPLEMENTED();
}
}
void FlowGraphCompiler::EmitNativeMoveArchitecture(
const compiler::ffi::NativeLocation& destination,
const compiler::ffi::NativeLocation& source) {
const auto& src_payload_type = source.payload_type();
const auto& dst_payload_type = destination.payload_type();
const auto& src_container_type = source.container_type();
const auto& dst_container_type = destination.container_type();
ASSERT(src_container_type.IsFloat() == dst_container_type.IsFloat());
ASSERT(src_container_type.IsInt() == dst_container_type.IsInt());
ASSERT(src_payload_type.IsSigned() == dst_payload_type.IsSigned());
ASSERT(src_payload_type.IsPrimitive());
ASSERT(dst_payload_type.IsPrimitive());
const intptr_t src_size = src_payload_type.SizeInBytes();
const intptr_t dst_size = dst_payload_type.SizeInBytes();
const bool sign_or_zero_extend = dst_size > src_size;
if (source.IsRegisters()) {
const auto& src = source.AsRegisters();
ASSERT(src.num_regs() == 1);
const auto src_reg = src.reg_at(0);
if (destination.IsRegisters()) {
const auto& dst = destination.AsRegisters();
ASSERT(dst.num_regs() == 1);
const auto dst_reg = dst.reg_at(0);
ASSERT(destination.container_type().SizeInBytes() <= 8);
if (!sign_or_zero_extend) {
__ MoveRegister(dst_reg, src_reg);
} else {
if (src_payload_type.IsSigned()) {
__ sbfx(dst_reg, src_reg, 0, src_size * kBitsPerByte);
} else {
__ ubfx(dst_reg, src_reg, 0, src_size * kBitsPerByte);
}
}
} else if (destination.IsFpuRegisters()) {
// Fpu Registers should only contain doubles and registers only ints.
UNIMPLEMENTED();
} else {
ASSERT(destination.IsStack());
const auto& dst = destination.AsStack();
ASSERT(!sign_or_zero_extend);
auto const op_size =
BytesToOperandSize(destination.container_type().SizeInBytes());
__ StoreToOffset(src.reg_at(0), dst.base_register(),
dst.offset_in_bytes(), op_size);
}
} else if (source.IsFpuRegisters()) {
const auto& src = source.AsFpuRegisters();
// We have not implemented conversions here, use IL convert instructions.
ASSERT(src_payload_type.Equals(dst_payload_type));
if (destination.IsRegisters()) {
// Fpu Registers should only contain doubles and registers only ints.
UNIMPLEMENTED();
} else if (destination.IsFpuRegisters()) {
const auto& dst = destination.AsFpuRegisters();
__ vmov(dst.fpu_reg(), src.fpu_reg());
} else {
ASSERT(destination.IsStack());
ASSERT(src_payload_type.IsFloat());
const auto& dst = destination.AsStack();
switch (dst_size) {
case 8:
__ StoreDToOffset(src.fpu_reg(), dst.base_register(),
dst.offset_in_bytes());
return;
case 4:
__ StoreSToOffset(src.fpu_reg(), dst.base_register(),
dst.offset_in_bytes());
return;
default:
UNREACHABLE();
}
}
} else {
ASSERT(source.IsStack());
const auto& src = source.AsStack();
if (destination.IsRegisters()) {
const auto& dst = destination.AsRegisters();
ASSERT(dst.num_regs() == 1);
const auto dst_reg = dst.reg_at(0);
EmitNativeLoad(dst_reg, src.base_register(), src.offset_in_bytes(),
src_payload_type.AsPrimitive().representation());
} else if (destination.IsFpuRegisters()) {
ASSERT(src_payload_type.Equals(dst_payload_type));
ASSERT(src_payload_type.IsFloat());
const auto& dst = destination.AsFpuRegisters();
switch (src_size) {
case 8:
__ LoadDFromOffset(dst.fpu_reg(), src.base_register(),
src.offset_in_bytes());
return;
case 4:
__ LoadSFromOffset(dst.fpu_reg(), src.base_register(),
src.offset_in_bytes());
return;
default:
UNIMPLEMENTED();
}
} else {
ASSERT(destination.IsStack());
UNREACHABLE();
}
}
}
void FlowGraphCompiler::EmitNativeLoad(Register dst,
Register base,
intptr_t offset,
compiler::ffi::PrimitiveType type) {
switch (type) {
case compiler::ffi::kInt8:
__ LoadFromOffset(dst, base, offset, compiler::kByte);
break;
case compiler::ffi::kUint8:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedByte);
break;
case compiler::ffi::kInt16:
__ LoadFromOffset(dst, base, offset, compiler::kTwoBytes);
break;
case compiler::ffi::kUint16:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedTwoBytes);
break;
case compiler::ffi::kInt32:
__ LoadFromOffset(dst, base, offset, compiler::kFourBytes);
break;
case compiler::ffi::kUint32:
case compiler::ffi::kFloat:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
break;
case compiler::ffi::kInt64:
case compiler::ffi::kUint64:
case compiler::ffi::kDouble:
__ LoadFromOffset(dst, base, offset, compiler::kEightBytes);
break;
case compiler::ffi::kInt24:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedTwoBytes);
__ LoadFromOffset(TMP, base, offset + 2, compiler::kByte);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 16));
break;
case compiler::ffi::kUint24:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedTwoBytes);
__ LoadFromOffset(TMP, base, offset + 2, compiler::kUnsignedByte);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 16));
break;
case compiler::ffi::kInt40:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
__ LoadFromOffset(TMP, base, offset + 4, compiler::kByte);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 32));
break;
case compiler::ffi::kUint40:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
__ LoadFromOffset(TMP, base, offset + 4, compiler::kUnsignedByte);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 32));
break;
case compiler::ffi::kInt48:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
__ LoadFromOffset(TMP, base, offset + 4, compiler::kTwoBytes);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 32));
break;
case compiler::ffi::kUint48:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
__ LoadFromOffset(TMP, base, offset + 4, compiler::kUnsignedTwoBytes);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 32));
break;
case compiler::ffi::kInt56:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
__ LoadFromOffset(TMP, base, offset + 4, compiler::kUnsignedTwoBytes);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 32));
__ LoadFromOffset(TMP, base, offset + 6, compiler::kByte);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 48));
break;
case compiler::ffi::kUint56:
__ LoadFromOffset(dst, base, offset, compiler::kUnsignedFourBytes);
__ LoadFromOffset(TMP, base, offset + 4, compiler::kUnsignedTwoBytes);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 32));
__ LoadFromOffset(TMP, base, offset + 6, compiler::kUnsignedByte);
__ orr(dst, dst, compiler::Operand(TMP, LSL, 48));
break;
default:
UNREACHABLE();
}
}
#undef __
#define __ compiler_->assembler()->
void ParallelMoveEmitter::EmitSwap(const MoveOperands& move) {
const Location source = move.src();
const Location destination = move.dest();
if (source.IsRegister() && destination.IsRegister()) {
ASSERT(source.reg() != TMP);
ASSERT(destination.reg() != TMP);
__ mov(TMP, source.reg());
__ mov(source.reg(), destination.reg());
__ mov(destination.reg(), TMP);
} else if (source.IsRegister() && destination.IsStackSlot()) {
Exchange(source.reg(), destination.base_reg(),
destination.ToStackSlotOffset());
} else if (source.IsStackSlot() && destination.IsRegister()) {
Exchange(destination.reg(), source.base_reg(), source.ToStackSlotOffset());
} else if (source.IsStackSlot() && destination.IsStackSlot()) {
Exchange(source.base_reg(), source.ToStackSlotOffset(),
destination.base_reg(), destination.ToStackSlotOffset());
} else if (source.IsFpuRegister() && destination.IsFpuRegister()) {
const VRegister dst = destination.fpu_reg();
const VRegister src = source.fpu_reg();
__ vmov(VTMP, src);
__ vmov(src, dst);
__ vmov(dst, VTMP);
} else if (source.IsFpuRegister() || destination.IsFpuRegister()) {
ASSERT(destination.IsDoubleStackSlot() || destination.IsQuadStackSlot() ||
source.IsDoubleStackSlot() || source.IsQuadStackSlot());
bool double_width =
destination.IsDoubleStackSlot() || source.IsDoubleStackSlot();
VRegister reg =
source.IsFpuRegister() ? source.fpu_reg() : destination.fpu_reg();
Register base_reg =
source.IsFpuRegister() ? destination.base_reg() : source.base_reg();
const intptr_t slot_offset = source.IsFpuRegister()
? destination.ToStackSlotOffset()
: source.ToStackSlotOffset();
if (double_width) {
__ LoadDFromOffset(VTMP, base_reg, slot_offset);
__ StoreDToOffset(reg, base_reg, slot_offset);
__ fmovdd(reg, VTMP);
} else {
__ LoadQFromOffset(VTMP, base_reg, slot_offset);
__ StoreQToOffset(reg, base_reg, slot_offset);
__ vmov(reg, VTMP);
}
} else if (source.IsDoubleStackSlot() && destination.IsDoubleStackSlot()) {
const intptr_t source_offset = source.ToStackSlotOffset();
const intptr_t dest_offset = destination.ToStackSlotOffset();
ScratchFpuRegisterScope ensure_scratch(this, kNoFpuRegister);
VRegister scratch = ensure_scratch.reg();
__ LoadDFromOffset(VTMP, source.base_reg(), source_offset);
__ LoadDFromOffset(scratch, destination.base_reg(), dest_offset);
__ StoreDToOffset(VTMP, destination.base_reg(), dest_offset);
__ StoreDToOffset(scratch, source.base_reg(), source_offset);
} else if (source.IsQuadStackSlot() && destination.IsQuadStackSlot()) {
const intptr_t source_offset = source.ToStackSlotOffset();
const intptr_t dest_offset = destination.ToStackSlotOffset();
ScratchFpuRegisterScope ensure_scratch(this, kNoFpuRegister);
VRegister scratch = ensure_scratch.reg();
__ LoadQFromOffset(VTMP, source.base_reg(), source_offset);
__ LoadQFromOffset(scratch, destination.base_reg(), dest_offset);
__ StoreQToOffset(VTMP, destination.base_reg(), dest_offset);
__ StoreQToOffset(scratch, source.base_reg(), source_offset);
} else {
UNREACHABLE();
}
}
void ParallelMoveEmitter::MoveMemoryToMemory(const compiler::Address& dst,
const compiler::Address& src) {
UNREACHABLE();
}
// Do not call or implement this function. Instead, use the form below that
// uses an offset from the frame pointer instead of an Address.
void ParallelMoveEmitter::Exchange(Register reg, const compiler::Address& mem) {
UNREACHABLE();
}
// Do not call or implement this function. Instead, use the form below that
// uses offsets from the frame pointer instead of Addresses.
void ParallelMoveEmitter::Exchange(const compiler::Address& mem1,
const compiler::Address& mem2) {
UNREACHABLE();
}
void ParallelMoveEmitter::Exchange(Register reg,
Register base_reg,
intptr_t stack_offset) {
ScratchRegisterScope tmp(this, reg);
__ mov(tmp.reg(), reg);
__ LoadFromOffset(reg, base_reg, stack_offset);
__ StoreToOffset(tmp.reg(), base_reg, stack_offset);
}
void ParallelMoveEmitter::Exchange(Register base_reg1,
intptr_t stack_offset1,
Register base_reg2,
intptr_t stack_offset2) {
ScratchRegisterScope tmp1(this, kNoRegister);
ScratchRegisterScope tmp2(this, tmp1.reg());
__ LoadFromOffset(tmp1.reg(), base_reg1, stack_offset1);
__ LoadFromOffset(tmp2.reg(), base_reg2, stack_offset2);
__ StoreToOffset(tmp1.reg(), base_reg2, stack_offset2);
__ StoreToOffset(tmp2.reg(), base_reg1, stack_offset1);
}
void ParallelMoveEmitter::SpillScratch(Register reg) {
__ Push(reg);
}
void ParallelMoveEmitter::RestoreScratch(Register reg) {
__ Pop(reg);
}
void ParallelMoveEmitter::SpillFpuScratch(FpuRegister reg) {
__ PushQuad(reg);
}
void ParallelMoveEmitter::RestoreFpuScratch(FpuRegister reg) {
__ PopQuad(reg);
}
#undef __
} // namespace dart
#endif // defined(TARGET_ARCH_ARM64)