3e7cda8a4e
This CL adds passing structs by value in FFI trampolines.
Nested structs and inline arrays are future work.
C defines passing empty structs as undefined behavior, so that is not
supported in this CL.
Suggested review order:
1) commit message
2) ffi/marshaller (decisions for what is done in IL and what in MC)
3) frontend/kernel_to_il (IL construction)
4) backend/il (MC generation from IL)
5) rest in VM
Overall architecture is that structs are split up into word-size chunks
in IL when this is possible: 1 definition in IL per chunk, 1 Location in
IL per chunk, and 1 NativeLocation for the backend per chunk.
In some cases it is not possible or less convenient to split into
chunks. In these cases TypedDataBase objects are stored into and loaded
from directly in machine code.
The various cases:
- FFI call arguments which are not passed as pointers: pass individual
chunks to FFI call which already have the right location.
- FFI call arguments which are passed as pointers: Pass in TypedDataBase
to FFI call, allocate space on the stack, and make a copy on the stack
and pass the copies' address to the callee.
- FFI call return value: pass in TypedData to FFI call, and copy result
in machine code.
- FFI callback arguments which are not passed as pointers: IL definition
for each chunk, and populate a new TypedData with those chunks.
- FFI callback arguments which are passed as pointer: IL definition for
the pointer, and copying of contents in IL.
- FFI return value when location is pointer: Copy data to callee result
location in IL.
- FFI return value when location is not a pointer: Copy data in machine
code to the right registers.
Some other notes about the implementation:
- Due to Store/LoadIndexed loading doubles from float arrays, we use
a int32 instead and use the BitCastInstr.
- Linux ia32 uses `ret 4` when returning structs by value. This requires
special casing in the FFI callback trampolines to either use `ret` or
`ret 4` when returning.
- The 1 IL definition, 1 Location, and 1 NativeLocation approach does
not remove the need for special casing PairLocations in the machine
code generation because they are 1 Location belonging to 1 definition.
Because of the amount of corner cases in the calling conventions that
need to be covered, the tests are generated, rather than hand-written.
ABIs tested on CQ: x64 (Linux, MacOS, Windows), ia32 (Linux, Windows),
arm (Android softFP, Linux hardFP), arm64 Android.
ABIs tested locally through Flutter: ia32 Android (emulator), x64 iOS
(simulator), arm64 iOS.
ABIs not tested: arm iOS.
TEST=runtime/bin/ffi_test/ffi_test_functions_generated.cc
TEST=runtime/bin/ffi_test/ffi_test_functions.cc
TEST=tests/{ffi,ffi_2}/function_structs_by_value_generated_test.dart
TEST=tests/{ffi,ffi_2}/function_callbacks_structs_by_value_generated_tes
TEST=tests/{ffi,ffi_2}/function_callbacks_structs_by_value_test.dart
TEST=tests/{ffi,ffi_2}/vmspecific_static_checks_test.dart
Closes https://github.com/dart-lang/sdk/issues/36730.
Change-Id: I474d3a4ee1faadbe767ddadd1b696e24d8dc364c
Cq-Include-Trybots: luci.dart.try:dart-sdk-linux-try,dart-sdk-mac-try,dart-sdk-win-try,vm-ffi-android-debug-arm-try,vm-ffi-android-debug-arm64-try,vm-kernel-asan-linux-release-x64-try,vm-kernel-mac-debug-x64-try,vm-kernel-linux-debug-ia32-try,vm-kernel-linux-debug-x64-try,vm-kernel-nnbd-linux-debug-x64-try,vm-kernel-nnbd-linux-debug-ia32-try,vm-kernel-nnbd-mac-release-x64-try,vm-kernel-nnbd-win-debug-x64-try,vm-kernel-precomp-linux-debug-x64-try,vm-kernel-precomp-linux-debug-simarm_x64-try,vm-kernel-precomp-nnbd-linux-debug-x64-try,vm-kernel-precomp-win-release-x64-try,vm-kernel-reload-linux-debug-x64-try,vm-kernel-reload-rollback-linux-debug-x64-try,vm-kernel-win-debug-x64-try,vm-kernel-win-debug-ia32-try,vm-precomp-ffi-qemu-linux-release-arm-try,vm-kernel-precomp-obfuscate-linux-release-x64-try,vm-kernel-msan-linux-release-x64-try,vm-kernel-precomp-msan-linux-release-x64-try,vm-kernel-precomp-android-release-arm_x64-try,analyzer-analysis-server-linux-try
Reviewed-on: https://dart-review.googlesource.com/c/sdk/+/140290
Commit-Queue: Daco Harkes <dacoharkes@google.com>
Reviewed-by: Martin Kustermann <kustermann@google.com>
190 lines
6.4 KiB
C++
190 lines
6.4 KiB
C++
// Copyright (c) 2019, the Dart project authors. Please see the AUTHORS file
|
|
// for details. All rights reserved. Use of this source code is governed by a
|
|
// BSD-style license that can be found in the LICENSE file.
|
|
|
|
#ifndef RUNTIME_VM_COMPILER_STUB_CODE_COMPILER_H_
|
|
#define RUNTIME_VM_COMPILER_STUB_CODE_COMPILER_H_
|
|
|
|
#if defined(DART_PRECOMPILED_RUNTIME)
|
|
#error "AOT runtime should not use compiler sources (including header files)"
|
|
#endif // defined(DART_PRECOMPILED_RUNTIME)
|
|
|
|
#include <functional>
|
|
|
|
#include "vm/allocation.h"
|
|
#include "vm/compiler/runtime_api.h"
|
|
#include "vm/constants.h"
|
|
#include "vm/growable_array.h"
|
|
#include "vm/stub_code_list.h"
|
|
#include "vm/tagged_pointer.h"
|
|
|
|
namespace dart {
|
|
|
|
// Forward declarations.
|
|
class Code;
|
|
|
|
namespace compiler {
|
|
|
|
// Forward declarations.
|
|
class Assembler;
|
|
|
|
// Represents an unresolved PC-relative Call/TailCall.
|
|
class UnresolvedPcRelativeCall : public ZoneAllocated {
|
|
public:
|
|
UnresolvedPcRelativeCall(intptr_t offset,
|
|
const dart::Code& target,
|
|
bool is_tail_call)
|
|
: offset_(offset), target_(target), is_tail_call_(is_tail_call) {}
|
|
|
|
intptr_t offset() const { return offset_; }
|
|
const dart::Code& target() const { return target_; }
|
|
bool is_tail_call() const { return is_tail_call_; }
|
|
|
|
private:
|
|
const intptr_t offset_;
|
|
const dart::Code& target_;
|
|
const bool is_tail_call_;
|
|
};
|
|
|
|
using UnresolvedPcRelativeCalls = GrowableArray<UnresolvedPcRelativeCall*>;
|
|
|
|
class StubCodeCompiler : public AllStatic {
|
|
public:
|
|
#if !defined(TARGET_ARCH_IA32)
|
|
static void GenerateBuildMethodExtractorStub(
|
|
Assembler* assembler,
|
|
const Object& closure_allocation_stub,
|
|
const Object& context_allocation_stub);
|
|
#endif
|
|
|
|
static ArrayPtr BuildStaticCallsTable(
|
|
Zone* zone,
|
|
compiler::UnresolvedPcRelativeCalls* unresolved_calls);
|
|
|
|
#define STUB_CODE_GENERATE(name) \
|
|
static void Generate##name##Stub(Assembler* assembler);
|
|
VM_STUB_CODE_LIST(STUB_CODE_GENERATE)
|
|
#undef STUB_CODE_GENERATE
|
|
|
|
static void GenerateAllocationStubForClass(
|
|
Assembler* assembler,
|
|
UnresolvedPcRelativeCalls* unresolved_calls,
|
|
const Class& cls,
|
|
const dart::Code& allocate_object,
|
|
const dart::Code& allocat_object_parametrized);
|
|
|
|
enum Optimized {
|
|
kUnoptimized,
|
|
kOptimized,
|
|
};
|
|
enum CallType {
|
|
kInstanceCall,
|
|
kStaticCall,
|
|
};
|
|
enum Exactness {
|
|
kCheckExactness,
|
|
kIgnoreExactness,
|
|
};
|
|
static void GenerateNArgsCheckInlineCacheStub(
|
|
Assembler* assembler,
|
|
intptr_t num_args,
|
|
const RuntimeEntry& handle_ic_miss,
|
|
Token::Kind kind,
|
|
Optimized optimized,
|
|
CallType type,
|
|
Exactness exactness);
|
|
static void GenerateNArgsCheckInlineCacheStubForEntryKind(
|
|
Assembler* assembler,
|
|
intptr_t num_args,
|
|
const RuntimeEntry& handle_ic_miss,
|
|
Token::Kind kind,
|
|
Optimized optimized,
|
|
CallType type,
|
|
Exactness exactness,
|
|
CodeEntryKind entry_kind);
|
|
static void GenerateUsageCounterIncrement(Assembler* assembler,
|
|
Register temp_reg);
|
|
static void GenerateOptimizedUsageCounterIncrement(Assembler* assembler);
|
|
|
|
#if defined(TARGET_ARCH_X64)
|
|
static constexpr intptr_t kNativeCallbackTrampolineSize = 10;
|
|
static constexpr intptr_t kNativeCallbackSharedStubSize = 217;
|
|
static constexpr intptr_t kNativeCallbackTrampolineStackDelta = 2;
|
|
#elif defined(TARGET_ARCH_IA32)
|
|
static constexpr intptr_t kNativeCallbackTrampolineSize = 10;
|
|
static constexpr intptr_t kNativeCallbackSharedStubSize = 134;
|
|
static constexpr intptr_t kNativeCallbackTrampolineStackDelta = 4;
|
|
#elif defined(TARGET_ARCH_ARM)
|
|
static constexpr intptr_t kNativeCallbackTrampolineSize = 12;
|
|
static constexpr intptr_t kNativeCallbackSharedStubSize = 140;
|
|
static constexpr intptr_t kNativeCallbackTrampolineStackDelta = 4;
|
|
#elif defined(TARGET_ARCH_ARM64)
|
|
static constexpr intptr_t kNativeCallbackTrampolineSize = 12;
|
|
static constexpr intptr_t kNativeCallbackSharedStubSize = 268;
|
|
static constexpr intptr_t kNativeCallbackTrampolineStackDelta = 2;
|
|
#endif
|
|
|
|
static void GenerateJITCallbackTrampolines(Assembler* assembler,
|
|
intptr_t next_callback_id);
|
|
|
|
// Calculates the offset (in words) from FP to the provided [cpu_register].
|
|
//
|
|
// Assumes
|
|
// * all [kDartAvailableCpuRegs] followed by saved-PC, saved-FP were
|
|
// pushed on the stack
|
|
// * [cpu_register] is in [kDartAvailableCpuRegs]
|
|
//
|
|
// The intended use of this function is to find registers on the stack which
|
|
// were spilled in the
|
|
// `StubCode::*<stub-name>Shared{With,Without}FpuRegsStub()`
|
|
static intptr_t WordOffsetFromFpToCpuRegister(Register cpu_register);
|
|
|
|
private:
|
|
// Common function for generating InitLateInstanceField and
|
|
// InitLateFinalInstanceField stubs.
|
|
static void GenerateInitLateInstanceFieldStub(Assembler* assembler,
|
|
bool is_final);
|
|
|
|
// Common function for generating Allocate<TypedData>Array stubs.
|
|
static void GenerateAllocateTypedDataArrayStub(Assembler* assembler,
|
|
intptr_t cid);
|
|
|
|
static void GenerateSharedStubGeneric(
|
|
Assembler* assembler,
|
|
bool save_fpu_registers,
|
|
intptr_t self_code_stub_offset_from_thread,
|
|
bool allow_return,
|
|
std::function<void()> perform_runtime_call);
|
|
|
|
// Generates shared slow path stub which saves registers and calls
|
|
// [target] runtime entry.
|
|
// If [store_runtime_result_in_result_register], then stub puts result into
|
|
// SharedSlowPathStubABI::kResultReg.
|
|
static void GenerateSharedStub(
|
|
Assembler* assembler,
|
|
bool save_fpu_registers,
|
|
const RuntimeEntry* target,
|
|
intptr_t self_code_stub_offset_from_thread,
|
|
bool allow_return,
|
|
bool store_runtime_result_in_result_register = false);
|
|
|
|
static void GenerateLateInitializationError(Assembler* assembler,
|
|
bool with_fpu_regs);
|
|
|
|
static void GenerateRangeError(Assembler* assembler, bool with_fpu_regs);
|
|
};
|
|
|
|
} // namespace compiler
|
|
|
|
enum DeoptStubKind { kLazyDeoptFromReturn, kLazyDeoptFromThrow, kEagerDeopt };
|
|
|
|
// Zap value used to indicate unused CODE_REG in deopt.
|
|
static const uword kZapCodeReg = 0xf1f1f1f1;
|
|
|
|
// Zap value used to indicate unused return address in deopt.
|
|
static const uword kZapReturnAddress = 0xe1e1e1e1;
|
|
|
|
} // namespace dart
|
|
|
|
#endif // RUNTIME_VM_COMPILER_STUB_CODE_COMPILER_H_
|