diff --git a/eng/common/cross/toolchain.cmake b/eng/common/cross/toolchain.cmake index 70b71395e3ba72..530ab5aa860783 100644 --- a/eng/common/cross/toolchain.cmake +++ b/eng/common/cross/toolchain.cmake @@ -333,6 +333,59 @@ if(TARGET_ARCH_NAME MATCHES "^(arm|armel)$") if(TARGET_ARCH_NAME STREQUAL "armel") add_compile_options(-mfloat-abi=softfp) endif() +elseif(TARGET_ARCH_NAME STREQUAL "riscv64") + # The ISA string and the ABI have to be settled here rather than in + # configurecompiler.cmake alone: cmake compiles and links its own probes at + # project() time, and a probe built for a different ABI than the sysroot fails + # to link, so the compiler is reported as broken before the build starts. + # + # Both are unset by default, which leaves the toolchain's own defaults in + # place. They may also come from the environment, the way CROSS_ROOTFS, + # TARGET_BUILD_ARCH and TOOLCHAIN already do, so that a build driven through a + # superproject can select them without reaching the cmake command line. + if(DEFINED ENV{CLR_CMAKE_RISCV64_MARCH}) + set(CLR_CMAKE_RISCV64_MARCH "$ENV{CLR_CMAKE_RISCV64_MARCH}") + endif() + if(DEFINED ENV{CLR_CMAKE_RISCV64_MABI}) + set(CLR_CMAKE_RISCV64_MABI "$ENV{CLR_CMAKE_RISCV64_MABI}") + endif() + + set(_riscv64_isa_flags "") + if(DEFINED CLR_CMAKE_RISCV64_MARCH) + string(APPEND _riscv64_isa_flags " -march=${CLR_CMAKE_RISCV64_MARCH}") + endif() + if(DEFINED CLR_CMAKE_RISCV64_MABI) + string(APPEND _riscv64_isa_flags " -mabi=${CLR_CMAKE_RISCV64_MABI}") + endif() + + if(NOT _riscv64_isa_flags STREQUAL "") + # *_FLAGS_INIT rather than add_compile_options: a try_compile runs as its own + # project and does not inherit directory properties, so the probe would still + # be built for the compiler's default ABI and fail against the sysroot. + string(APPEND CMAKE_C_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_CXX_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_ASM_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_EXE_LINKER_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_SHARED_LINKER_FLAGS_INIT "${_riscv64_isa_flags}") + endif() + + # Without the A extension the compiler lowers C/C++ atomics to __atomic_* + # calls instead of emitting lr/sc, and those live in libatomic. Not every link + # in the tree passes -latomic on its own. + # + # Only the single-letter part of the ISA string counts: a multi-letter extension + # whose name contains "a" (Zba, for one) is not the A extension, and "g" is + # shorthand for imafd and so implies it. + if(DEFINED CLR_CMAKE_RISCV64_MARCH) + string(REGEX REPLACE "_.*$" "" _riscv_single_letter "${CLR_CMAKE_RISCV64_MARCH}") + string(REGEX REPLACE "^rv[0-9]+" "" _riscv_single_letter "${_riscv_single_letter}") + if(NOT _riscv_single_letter MATCHES "[ag]") + add_toolchain_linker_flag("-latomic") + endif() + endif() + + # persist variables across multiple try_compile passes + list(APPEND CMAKE_TRY_COMPILE_PLATFORM_VARIABLES CLR_CMAKE_RISCV64_MARCH CLR_CMAKE_RISCV64_MABI) elseif(TARGET_ARCH_NAME STREQUAL "s390x") add_compile_options("--target=${TOOLCHAIN}") elseif(TARGET_ARCH_NAME STREQUAL "x86") diff --git a/eng/native/configurecompiler.cmake b/eng/native/configurecompiler.cmake index b24f43e7d8cc42..52fb966fa2d42a 100644 --- a/eng/native/configurecompiler.cmake +++ b/eng/native/configurecompiler.cmake @@ -925,8 +925,50 @@ if(CLR_CMAKE_HOST_UNIX_ARMV6) endif(CLR_CMAKE_HOST_UNIX_ARMV6) if(CLR_CMAKE_HOST_UNIX_RISCV64) - add_compile_options(-march=rv64gc) - add_compile_options(-mabi=lp64d) + # The ISA string and the ABI the native runtime is built with. The defaults are + # the rv64gc/lp64d baseline, so an unconfigured build is unchanged. A target + # whose sysroot is built for a different ABI - a soft-float lp64 userspace, for + # instance - selects it here, the way armel selects -mfloat-abi=softfp above, + # instead of overriding the flags further down the command line. + # + # The float-ABI field of e_flags follows -mabi (not -march), so setting the ABI + # once here is what makes the whole native build agree with the sysroot; the + # linker rejects a mix, and it is also passed at link time so that the driver + # selects the matching CRT and builtins. + set(CLR_CMAKE_RISCV64_MARCH "rv64gc" CACHE STRING "RISC-V ISA string for the native runtime build") + set(CLR_CMAKE_RISCV64_MABI "lp64d" CACHE STRING "RISC-V ABI for the native runtime build") + + # Also settable from the environment, the way CLR_CC, ROOTFS_DIR and TOOLCHAIN + # already are: a build driven through a superproject cannot always reach the + # cmake command line of an individual repository. + if(DEFINED ENV{CLR_CMAKE_RISCV64_MARCH}) + set(CLR_CMAKE_RISCV64_MARCH "$ENV{CLR_CMAKE_RISCV64_MARCH}") + endif() + if(DEFINED ENV{CLR_CMAKE_RISCV64_MABI}) + set(CLR_CMAKE_RISCV64_MABI "$ENV{CLR_CMAKE_RISCV64_MABI}") + endif() + + # Decide whether the ISA string selects the A extension. Only the single-letter + # part counts: a multi-letter extension whose name contains "a" (Zba, for one) + # is not the A extension, and "g" is shorthand for imafd and so implies it. + string(REGEX REPLACE "_.*$" "" _riscv_single_letter "${CLR_CMAKE_RISCV64_MARCH}") + string(REGEX REPLACE "^rv[0-9]+" "" _riscv_single_letter "${_riscv_single_letter}") + if(_riscv_single_letter MATCHES "[ag]") + set(_riscv_has_a ON) + else() + set(_riscv_has_a OFF) + endif() + + # Without the A extension every __atomic_* call is "not lock-free" and clang + # warns; the runtime builds with -Werror, which would make that fatal. + if(NOT _riscv_has_a) + add_compile_options(-Wno-atomic-alignment) + endif() + + add_compile_options(-march=${CLR_CMAKE_RISCV64_MARCH}) + add_compile_options(-mabi=${CLR_CMAKE_RISCV64_MABI}) + add_link_options(-march=${CLR_CMAKE_RISCV64_MARCH}) + add_link_options(-mabi=${CLR_CMAKE_RISCV64_MABI}) endif(CLR_CMAKE_HOST_UNIX_RISCV64) if(CLR_CMAKE_HOST_UNIX_X86) diff --git a/src/coreclr/debug/di/riscv64/floatconversion.S b/src/coreclr/debug/di/riscv64/floatconversion.S index 138db0bc9dd243..1b8e02957d387e 100644 --- a/src/coreclr/debug/di/riscv64/floatconversion.S +++ b/src/coreclr/debug/di/riscv64/floatconversion.S @@ -7,6 +7,8 @@ // input: (in A0) the address of the ULONGLONG to be converted to a double // output: the double corresponding to the ULONGLONG input value LEAF_ENTRY FPFillR8, .TEXT +#if __riscv_flen >= 64 fld fa0, 0(a0) +#endif ret LEAF_END FPFillR8, .TEXT diff --git a/src/coreclr/inc/clrconfigvalues.h b/src/coreclr/inc/clrconfigvalues.h index 42c47d615168c7..1adbeb64429198 100644 --- a/src/coreclr/inc/clrconfigvalues.h +++ b/src/coreclr/inc/clrconfigvalues.h @@ -720,6 +720,7 @@ RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zba, W("EnableRiscV64 RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zbb, W("EnableRiscV64Zbb"), 1, "Allows RiscV64 Zbb hardware intrinsics to be disabled") RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zbs, W("EnableRiscV64Zbs"), 1, "Allows RiscV64 Zbs hardware intrinsics to be disabled") RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zicond, W("EnableRiscV64Zicond"), 1, "Allows RiscV64 Zicond hardware intrinsics to be disabled") +RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Compressed, W("EnableRiscV64Compressed"), 1, "Allows the RiscV64 C (compressed) instruction set to be disabled") #endif /// diff --git a/src/coreclr/inc/corinfo.h b/src/coreclr/inc/corinfo.h index cd16f8abbbd3d9..c11c5dddcc69bc 100644 --- a/src/coreclr/inc/corinfo.h +++ b/src/coreclr/inc/corinfo.h @@ -342,6 +342,28 @@ enum CorInfoHelpFunc CORINFO_HELP_FLTREM, CORINFO_HELP_DBLREM, + /* Soft-float helpers, for targets without a floating-point unit (RISC-V + without the F/D extensions). The arithmetic helpers have the same + semantics as the corresponding IL instructions. The compare helpers + return a three-way result (< 0, 0, > 0 for less, equal, greater) and + differ only for unordered operands: CMP_LE returns 1 and CMP_GE returns + -1 (the libgcc __le*f2/__ge*f2 conventions), so that every ordered and + unordered IL comparison maps onto a single call. */ + CORINFO_HELP_FLTADD, + CORINFO_HELP_FLTSUB, + CORINFO_HELP_FLTMUL, + CORINFO_HELP_FLTDIV, + CORINFO_HELP_DBLADD, + CORINFO_HELP_DBLSUB, + CORINFO_HELP_DBLMUL, + CORINFO_HELP_DBLDIV, + CORINFO_HELP_FLTCMP_LE, + CORINFO_HELP_FLTCMP_GE, + CORINFO_HELP_DBLCMP_LE, + CORINFO_HELP_DBLCMP_GE, + CORINFO_HELP_FLT2DBL, + CORINFO_HELP_DBL2FLT, + /* Allocating a new object. Always use ICorClassInfo::getNewHelper() to decide which is the right helper to use to allocate an object of a given type. */ diff --git a/src/coreclr/inc/corinfoinstructionset.h b/src/coreclr/inc/corinfoinstructionset.h index 2f347572ee34d4..027c542a7699a1 100644 --- a/src/coreclr/inc/corinfoinstructionset.h +++ b/src/coreclr/inc/corinfoinstructionset.h @@ -65,6 +65,10 @@ enum CORINFO_InstructionSet InstructionSet_Zbb=3, InstructionSet_Zbs=4, InstructionSet_Zicond=5, + InstructionSet_F=6, + InstructionSet_D=7, + InstructionSet_C=8, + InstructionSet_A=9, #endif // TARGET_RISCV64 #ifdef TARGET_WASM InstructionSet_WasmBase=1, @@ -469,6 +473,14 @@ inline CORINFO_InstructionSetFlags EnsureInstructionSetFlagsAreValid(CORINFO_Ins resultflags.RemoveInstructionSet(InstructionSet_Zbs); if (resultflags.HasInstructionSet(InstructionSet_Zicond) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) resultflags.RemoveInstructionSet(InstructionSet_Zicond); + if (resultflags.HasInstructionSet(InstructionSet_F) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) + resultflags.RemoveInstructionSet(InstructionSet_F); + if (resultflags.HasInstructionSet(InstructionSet_D) && !resultflags.HasInstructionSet(InstructionSet_F)) + resultflags.RemoveInstructionSet(InstructionSet_D); + if (resultflags.HasInstructionSet(InstructionSet_C) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) + resultflags.RemoveInstructionSet(InstructionSet_C); + if (resultflags.HasInstructionSet(InstructionSet_A) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) + resultflags.RemoveInstructionSet(InstructionSet_A); #endif // TARGET_RISCV64 #ifdef TARGET_WASM if (resultflags.HasInstructionSet(InstructionSet_Vector128) && !resultflags.HasInstructionSet(InstructionSet_PackedSimd)) @@ -777,6 +789,14 @@ inline const char *InstructionSetToString(CORINFO_InstructionSet instructionSet) return "Zbs"; case InstructionSet_Zicond : return "Zicond"; + case InstructionSet_F : + return "F"; + case InstructionSet_D : + return "D"; + case InstructionSet_C : + return "C"; + case InstructionSet_A : + return "A"; #endif // TARGET_RISCV64 #ifdef TARGET_WASM case InstructionSet_WasmBase : @@ -989,6 +1009,10 @@ inline CORINFO_InstructionSet InstructionSetFromR2RInstructionSet(ReadyToRunInst case READYTORUN_INSTRUCTION_Zbb: return InstructionSet_Zbb; case READYTORUN_INSTRUCTION_Zbs: return InstructionSet_Zbs; case READYTORUN_INSTRUCTION_Zicond: return InstructionSet_Zicond; + case READYTORUN_INSTRUCTION_RiscV64F: return InstructionSet_F; + case READYTORUN_INSTRUCTION_RiscV64D: return InstructionSet_D; + case READYTORUN_INSTRUCTION_RiscV64C: return InstructionSet_C; + case READYTORUN_INSTRUCTION_RiscV64A: return InstructionSet_A; #endif // TARGET_RISCV64 #ifdef TARGET_WASM case READYTORUN_INSTRUCTION_WasmBase: return InstructionSet_WasmBase; diff --git a/src/coreclr/inc/corjitflags.h b/src/coreclr/inc/corjitflags.h index d898fcda5a2e59..34d9338997366a 100644 --- a/src/coreclr/inc/corjitflags.h +++ b/src/coreclr/inc/corjitflags.h @@ -61,7 +61,9 @@ class CORJIT_FLAGS #if defined(TARGET_ARM) CORJIT_FLAG_RELATIVE_CODE_RELOCS = 29, // JIT should generate PC-relative address computations instead of EE relocation records - CORJIT_FLAG_SOFTFP_ABI = 30, // Enable armel calling convention +#endif +#if defined(TARGET_ARM) || defined(TARGET_RISCV64) + CORJIT_FLAG_SOFTFP_ABI = 30, // Enable the soft-float calling convention (armel; lp64 on RISC-V) #endif CORJIT_FLAG_USE_DISPATCH_HELPERS = 31, // The JIT should use helpers for interface dispatch instead of virtual stub dispatch CORJIT_FLAG_VERIFY_GC_MODE_TRANSITIONS = 32, // The JIT should emit the diagnostic helpers that verify GC mode transitions are legal diff --git a/src/coreclr/inc/jiteeversionguid.h b/src/coreclr/inc/jiteeversionguid.h index fb33de9eaa0023..8f00452418b9f8 100644 --- a/src/coreclr/inc/jiteeversionguid.h +++ b/src/coreclr/inc/jiteeversionguid.h @@ -37,11 +37,11 @@ #include -constexpr GUID JITEEVersionIdentifier = { /* fa0c6a6f-b219-4b60-b928-042c72667eb3 */ - 0xfa0c6a6f, - 0xb219, - 0x4b60, - {0xb9, 0x28, 0x04, 0x2c, 0x72, 0x66, 0x7e, 0xb3} +constexpr GUID JITEEVersionIdentifier = { /* a3daece5-c930-442e-886d-0cf93f4d6e54 */ + 0xa3daece5, + 0xc930, + 0x442e, + {0x88, 0x6d, 0x0c, 0xf9, 0x3f, 0x4d, 0x6e, 0x54} }; #endif // JIT_EE_VERSIONING_GUID_H diff --git a/src/coreclr/inc/jithelpers.h b/src/coreclr/inc/jithelpers.h index 99a84286242c8f..bad88c5fab87f5 100644 --- a/src/coreclr/inc/jithelpers.h +++ b/src/coreclr/inc/jithelpers.h @@ -97,6 +97,24 @@ JITHELPER(CORINFO_HELP_FLTREM, JIT_FltRem, METHOD__NIL) JITHELPER(CORINFO_HELP_DBLREM, JIT_DblRem, METHOD__NIL) + // Soft-float helpers. The JIT only requests them under CORJIT_FLAG_SOFTFP_ABI on + // targets without an FPU (NativeAOT for RISC-V without F/D, where the AOT + // compiler binds them to the compiler-rt builtins); the VM never sets that flag. + JITHELPER(CORINFO_HELP_FLTADD, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTSUB, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTMUL, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTDIV, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLADD, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLSUB, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLMUL, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLDIV, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTCMP_LE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTCMP_GE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLCMP_LE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLCMP_GE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLT2DBL, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBL2FLT, NULL, METHOD__NIL) + // Allocating a new object JITHELPER(CORINFO_HELP_NEWFAST, RhpNew, METHOD__NIL) JITHELPER(CORINFO_HELP_NEWFAST_MAYBEFROZEN, RhpNewMaybeFrozen, METHOD__NIL) diff --git a/src/coreclr/inc/readytoruninstructionset.h b/src/coreclr/inc/readytoruninstructionset.h index d2851e91577f1e..96121c5d4f843c 100644 --- a/src/coreclr/inc/readytoruninstructionset.h +++ b/src/coreclr/inc/readytoruninstructionset.h @@ -103,6 +103,10 @@ enum ReadyToRunInstructionSet READYTORUN_INSTRUCTION_Cssc=93, READYTORUN_INSTRUCTION_Zicond=94, READYTORUN_INSTRUCTION_Fp16=95, + READYTORUN_INSTRUCTION_RiscV64F=96, + READYTORUN_INSTRUCTION_RiscV64D=97, + READYTORUN_INSTRUCTION_RiscV64C=98, + READYTORUN_INSTRUCTION_RiscV64A=99, }; diff --git a/src/coreclr/jit/codegencommon.cpp b/src/coreclr/jit/codegencommon.cpp index 8ea9c5f49c7850..29ca4d2fe39bde 100644 --- a/src/coreclr/jit/codegencommon.cpp +++ b/src/coreclr/jit/codegencommon.cpp @@ -8571,6 +8571,15 @@ void CodeGen::genPoisonFrame(regMaskTP regLiveIn) // void CodeGen::genBitCast(var_types targetType, regNumber targetReg, var_types srcType, regNumber srcReg) { +#ifdef TARGET_RISCV64 + if (m_compiler->opts.compUseSoftFP && (srcType == TYP_FLOAT) && (targetType == TYP_INT)) + { + // Soft-float: a float lives in an integer register with unspecified upper + // bits (RISC-V psABI), while an int is expected to be sign-extended. + GetEmitter()->emitIns_R_R_I(INS_addiw, EA_4BYTE, targetReg, srcReg, 0); + return; + } +#endif // TARGET_RISCV64 const bool srcFltReg = varTypeUsesFloatReg(srcType); assert(srcFltReg == genIsValidFloatReg(srcReg)); diff --git a/src/coreclr/jit/codegenriscv64.cpp b/src/coreclr/jit/codegenriscv64.cpp index aa4abea743baca..d85c8a6d338851 100644 --- a/src/coreclr/jit/codegenriscv64.cpp +++ b/src/coreclr/jit/codegenriscv64.cpp @@ -1008,6 +1008,26 @@ void CodeGen::genSetRegToConst(regNumber targetReg, var_types targetType, GenTre emitAttr size = emitActualTypeSize(tree); double constValue = tree->AsDblCon()->DconValue(); + if (m_compiler->opts.compUseSoftFP) + { + // No FP registers: the constant is its IEEE 754 bit pattern in an integer register. + assert(genIsValidIntReg(targetReg)); + int64_t bits; + if (size == EA_4BYTE) + { + float fltValue = (float)constValue; + int32_t fltBits; + memcpy(&fltBits, &fltValue, sizeof(fltBits)); + bits = fltBits; + } + else + { + memcpy(&bits, &constValue, sizeof(bits)); + } + instGen_Set_Reg_To_Imm(size, targetReg, bits); + break; + } + assert(emitter::isFloatReg(targetReg)); int64_t bits; if (emitter::isSingleInstructionFpImm(constValue, size, &bits)) @@ -2187,7 +2207,55 @@ void CodeGen::genLockedInstructions(GenTreeOp* treeNode) default: noway_assert(!"Unexpected treeNode->gtOper"); } - GetEmitter()->emitIns_R_R_R(ins, dataSize, targetReg, addrReg, dataReg); + if (!m_compiler->compOpportunisticallyDependsOn(InstructionSet_A)) + { + // Without the A extension the ISA has no atomic memory operation, so the + // only lowering available is a plain read/modify/write. That is correct + // only on a target with a single hart and no preemption, which is the + // condition under which a build may select an ISA without A. + // + // XCHG needs no scratch; the arithmetic and bitwise forms compute the new + // value in an internal register so the original stays live in targetReg. + // BuildNode in lsrariscv64 extends the address and data lifetimes so they + // are not reused across the sequence. + instruction insLoad = is4 ? INS_lw : INS_ld; + instruction insStore = is4 ? INS_sw : INS_sd; + if (treeNode->OperIs(GT_XCHG)) + { + if (targetReg != REG_ZERO) + { + GetEmitter()->emitIns_R_R_I(insLoad, dataSize, targetReg, addrReg, 0); + } + GetEmitter()->emitIns_R_R_I(insStore, dataSize, dataReg, addrReg, 0); + } + else + { + regNumber tmpReg = internalRegisters.GetSingle(treeNode); + regNumber valueReg = (targetReg != REG_ZERO) ? targetReg : tmpReg; + instruction insOp; + switch (treeNode->gtOper) + { + case GT_XADD: + insOp = is4 ? INS_addw : INS_add; + break; + case GT_XAND: + insOp = INS_and; + break; + case GT_XORR: + insOp = INS_or; + break; + default: + unreached(); + } + GetEmitter()->emitIns_R_R_I(insLoad, dataSize, valueReg, addrReg, 0); + GetEmitter()->emitIns_R_R_R(insOp, dataSize, tmpReg, valueReg, dataReg); + GetEmitter()->emitIns_R_R_I(insStore, dataSize, tmpReg, addrReg, 0); + } + } + else + { + GetEmitter()->emitIns_R_R_R(ins, dataSize, targetReg, addrReg, dataReg); + } if (targetReg != REG_ZERO) { @@ -2250,19 +2318,32 @@ void CodeGen::genCodeForCmpXchg(GenTreeCmpXchg* treeNode) // so mark the location register as a GC pointer until code generation for this node is finished. gcInfo.gcMarkRegPtrVal(loc, locOp->TypeGet()); - BasicBlock* retry = genCreateTempLabel(); - BasicBlock* fail = genCreateTempLabel(); + BasicBlock* fail = genCreateTempLabel(); emitter* e = GetEmitter(); emitAttr size = emitActualTypeSize(valOp); bool is4 = (size == EA_4BYTE); - genDefineTempLabel(retry); - e->emitIns_R_R_R(is4 ? INS_lr_w : INS_lr_d, size, target, loc, REG_R0); // load original value - e->emitIns_J_cond_la(INS_bne, fail, target, comparand); // fail if doesn’t match - e->emitIns_R_R_R(is4 ? INS_sc_w : INS_sc_d, size, storeErr, loc, val); // try to update - e->emitIns_J_cond_la(INS_bnez, retry, storeErr); // retry if update failed - genDefineTempLabel(fail); + if (!m_compiler->compOpportunisticallyDependsOn(InstructionSet_A)) + { + // No A extension, so no lr/sc reservation pair. Compare and swap without + // one: old = *loc; if (old == comparand) *loc = val; result = old. As + // above, this holds only on a single-hart target with no preemption. + e->emitIns_R_R_I(is4 ? INS_lw : INS_ld, size, target, loc, 0); + e->emitIns_J_cond_la(INS_bne, fail, target, comparand); + e->emitIns_R_R_I(is4 ? INS_sw : INS_sd, size, val, loc, 0); + genDefineTempLabel(fail); + } + else + { + BasicBlock* retry = genCreateTempLabel(); + genDefineTempLabel(retry); + e->emitIns_R_R_R(is4 ? INS_lr_w : INS_lr_d, size, target, loc, REG_R0); // load original value + e->emitIns_J_cond_la(INS_bne, fail, target, comparand); // fail if doesn't match + e->emitIns_R_R_R(is4 ? INS_sc_w : INS_sc_d, size, storeErr, loc, val); // try to update + e->emitIns_J_cond_la(INS_bnez, retry, storeErr); // retry if update failed + genDefineTempLabel(fail); + } gcInfo.gcMarkRegSetNpt(locOp->gtGetRegMask()); genProduceReg(treeNode); diff --git a/src/coreclr/jit/compiler.cpp b/src/coreclr/jit/compiler.cpp index b16af4edf9c91f..45c6f6ecff3f9a 100644 --- a/src/coreclr/jit/compiler.cpp +++ b/src/coreclr/jit/compiler.cpp @@ -53,6 +53,9 @@ MethodSet* Compiler::s_pJitMethodSet = nullptr; bool GlobalJitOptions::compFeatureHfa = false; LONG GlobalJitOptions::compUseSoftFPConfigured = 0; #endif // CONFIGURABLE_ARM_ABI +#ifdef TARGET_RISCV64 +LONG GlobalJitOptions::compUseSoftFPConfigured = 0; +#endif // TARGET_RISCV64 /***************************************************************************** * @@ -867,7 +870,7 @@ var_types Compiler::getReturnTypeForStruct(CORINFO_CLASS_HANDLE clsHnd, useType = TYP_UNKNOWN; } #elif defined(TARGET_RISCV64) || defined(TARGET_LOONGARCH64) - if (structSize <= (TARGET_POINTER_SIZE * 2)) + if ((structSize <= (TARGET_POINTER_SIZE * 2)) && !opts.compUseSoftFP) { const CORINFO_FPSTRUCT_LOWERING* lowering = GetFpStructLowering(clsHnd); if (!lowering->byIntegerCallConv) @@ -1941,6 +1944,20 @@ void Compiler::compSetProcessor() // Add virtual vector ISA. Vector128 is part of the required Wasm SIMD baseline. instructionSetFlags.AddInstructionSet(InstructionSet_Vector128); +#elif defined(TARGET_RISCV64) + // Ensure the required baseline ISA is supported in JIT code, even if not passed in by the VM. + instructionSetFlags.AddInstructionSet(InstructionSet_RiscV64Base); + + // C is the one base extension with a config opt-out: turning it off only stops + // the JIT from emitting compressed encodings, so it is safe to honor here however + // the flags were seeded. F, D and A have no opt-out; a target without them is + // selected through the AOT compiler's instruction set. + if (JitConfig.EnableRiscV64Compressed() == 0) + { + instructionSetFlags.RemoveInstructionSet(InstructionSet_C); + } + + instructionSetFlags = EnsureInstructionSetFlagsAreValid(instructionSetFlags); #endif // TARGET_ARM64 assert(instructionSetFlags.Equals(EnsureInstructionSetFlagsAreValid(instructionSetFlags))); @@ -2488,6 +2505,11 @@ void Compiler::compInitOptions(JitFlags* jitFlags) if (compIsForInlining()) { +#ifdef TARGET_RISCV64 + // The soft-float mode is decided by the root compilation (see below); the + // importer of an inlinee needs it too, for the intrinsics and casts it expands. + opts.compUseSoftFP = impInlineInfo->InlinerCompiler->opts.compUseSoftFP; +#endif // TARGET_RISCV64 return; } @@ -2892,6 +2914,51 @@ void Compiler::compInitOptions(JitFlags* jitFlags) } GlobalJitOptions::compFeatureHfa = !opts.compUseSoftFP; +#elif defined(TARGET_RISCV64) + // Soft-float, for targets without the F/D extensions (set by the AOT driver): + // the lp64 calling convention passes FP values in integer registers, and + // TYP_FLOAT/TYP_DOUBLE values live in the integer register file altogether; + // the FP arithmetic is done by helper calls (see fgMorphSmpOp). The register + // class of a type is a process-wide table, so the setting cannot change + // during the lifetime of the process. + opts.compUseSoftFP = jitFlags->IsSet(JitFlags::JIT_FLAG_SOFTFP_ABI); + + // The first compilation of the process fixes the mode: it claims the + // configuration, initializes the table and then publishes the mode. Every + // other compilation waits for the publication and must request the same + // mode, so the table is never written while another compilation may read it. + enum SoftFPConfig : LONG + { + SoftFPConfigUnset = 0, + SoftFPConfigHard = 1, + SoftFPConfigSoft = 2, + SoftFPConfigInitializing = 3, + }; + const LONG softFPConfig = opts.compUseSoftFP ? SoftFPConfigSoft : SoftFPConfigHard; + LONG oldSoftFPConfig = InterlockedCompareExchange(&GlobalJitOptions::compUseSoftFPConfigured, + SoftFPConfigInitializing, SoftFPConfigUnset); + if (oldSoftFPConfig == SoftFPConfigUnset) + { + if (opts.compUseSoftFP) + { + varTypeRegister[TYP_FLOAT] = VTR_INT; + varTypeRegister[TYP_DOUBLE] = VTR_INT; + } + InterlockedExchange(&GlobalJitOptions::compUseSoftFPConfigured, softFPConfig); + } + else + { + while (oldSoftFPConfig == SoftFPConfigInitializing) + { + // Atomic read; the initialization window is two byte stores long. + oldSoftFPConfig = InterlockedCompareExchange(&GlobalJitOptions::compUseSoftFPConfigured, SoftFPConfigUnset, + SoftFPConfigUnset); + } + if (oldSoftFPConfig != softFPConfig) + { + NO_WAY("SoftFP setting changed during lifetime of process"); + } + } #elif defined(ARM_SOFTFP) && defined(TARGET_ARM) // Armel is unconditionally enabled in the JIT. Verify that the VM side agrees. assert(jitFlags->IsSet(JitFlags::JIT_FLAG_SOFTFP_ABI)); @@ -6245,6 +6312,18 @@ int Compiler::compCompileAfterInit(CORINFO_MODULE_HANDLE classPtr, { instructionSetFlags.AddInstructionSet(InstructionSet_Zicond); } + + // F, D and A are part of the rv64gc baseline and cannot be turned off here: a target + // without them is selected through the AOT compiler's instruction set, which also + // checks that the ABI and the execution environment allow it. + instructionSetFlags.AddInstructionSet(InstructionSet_F); + instructionSetFlags.AddInstructionSet(InstructionSet_D); + instructionSetFlags.AddInstructionSet(InstructionSet_A); + + if (JitConfig.EnableRiscV64Compressed() != 0) + { + instructionSetFlags.AddInstructionSet(InstructionSet_C); + } #endif // These calls are important and explicitly ordered to ensure that the flags are correct in diff --git a/src/coreclr/jit/compiler.h b/src/coreclr/jit/compiler.h index c09c3f8cb8b4b9..ec793af1368e7f 100644 --- a/src/coreclr/jit/compiler.h +++ b/src/coreclr/jit/compiler.h @@ -7285,6 +7285,15 @@ class Compiler bool fgIsBlockCold(BasicBlock* block); GenTree* fgMorphCastIntoHelper(GenTree* tree, int helper, GenTree* oper); +#ifdef TARGET_RISCV64 + // Soft-float expansion of the FP operations into helper calls + GenTree* fgMorphSoftFloatArith(GenTreeOp* tree); + GenTree* fgMorphSoftFloatCast(GenTreeCast* tree, CorInfoHelpFunc helper, GenTree* oper); + GenTree* fgMorphSoftFloatNeg(GenTreeOp* neg); + GenTree* fgMorphSoftFloatRelop(GenTreeOp* relop); + GenTree* fgMorphSoftFloatCkFinite(GenTreeOp* ckFinite); + GenTree* fgMorphSoftFloatCastToInt32(GenTree* src, bool toUnsigned); +#endif // TARGET_RISCV64 GenTree* fgMorphIntoHelperCall( GenTree* tree, int helper, bool morphArgs, GenTree* arg1 = nullptr, GenTree* arg2 = nullptr); @@ -10960,7 +10969,7 @@ class Compiler // support/nonsupport for an instruction set bool compIsaSupportedDebugOnly(CORINFO_InstructionSet isa) const { -#if defined(TARGET_XARCH) || defined(TARGET_ARM64) +#if defined(TARGET_XARCH) || defined(TARGET_ARM64) || defined(TARGET_RISCV64) return opts.compSupportsISA.HasInstructionSet(isa); #else return false; @@ -11653,7 +11662,9 @@ class Compiler int compJitSaveFpLrWithCalleeSavedRegisters; #endif // defined(TARGET_ARM64) -#ifdef CONFIGURABLE_ARM_ABI +#if defined(CONFIGURABLE_ARM_ABI) || defined(TARGET_RISCV64) + // On RISCV64 the lp64 soft-float ABI is selected per compilation by + // JIT_FLAG_SOFTFP_ABI (no-F targets); see compInitOptions. bool compUseSoftFP = false; #else #ifdef ARM_SOFTFP diff --git a/src/coreclr/jit/emitriscv64.cpp b/src/coreclr/jit/emitriscv64.cpp index 9dcd3a29d7aaab..3620f01a4c30fc 100644 --- a/src/coreclr/jit/emitriscv64.cpp +++ b/src/coreclr/jit/emitriscv64.cpp @@ -1012,6 +1012,13 @@ void emitter::emitIns_R_R_R( bool emitter::tryEmitCompressedIns_R_R_R( instruction ins, emitAttr attr, regNumber rd, regNumber rs1, regNumber rs2, insOpts opt) { + // Targets without the C extension (e.g. a zkVM guest on rv64im) must never + // receive a compressed encoding. This is the only place the JIT emits one. + if (!m_compiler->compOpportunisticallyDependsOn(InstructionSet_C)) + { + return false; + } + // TODO-RISCV64-RVC: Disable this early return once compresed instructions are allowed in prolog / epilog if (emitGeneratingPrologOrFuncletProlog() || emitGeneratingEpilogOrFuncletEpilog()) { @@ -2191,6 +2198,29 @@ unsigned emitter::emitOutput_Instr(BYTE* dst, code_t code) const { assert(dst != nullptr); static_assert(sizeof(code_t) == 4, "code_t must be 4 bytes"); +#ifdef DEBUG + // On rv64 these major opcodes decode exclusively to F/D-extension + // instructions, so a no-F target must never emit one. The compressed FP + // forms are unreachable once C is gated off; checked for completeness. + switch (GetMajorOpcode(code)) + { + case MajorOpcode::LoadFp: + case MajorOpcode::StoreFp: + case MajorOpcode::MAdd: + case MajorOpcode::MSub: + case MajorOpcode::NmSub: + case MajorOpcode::NmAdd: + case MajorOpcode::OpFp: + case MajorOpcode::Fld: + case MajorOpcode::Fsd: + case MajorOpcode::FldSp: + case MajorOpcode::FsdSp: + assert(m_compiler->compIsaSupportedDebugOnly(InstructionSet_F)); + break; + default: + break; + } +#endif // DEBUG unsigned codeSize = Is32BitInstruction((WORD)code) ? 4 : 2; assert((codeSize == 4) || ((code >> 16) == 0)); memcpy(dst + writeableOffset, &code, codeSize); @@ -4964,7 +4994,7 @@ void emitter::emitInsLoadStoreOp(instruction ins, emitAttr attr, regNumber dataR } else { - bool needTemp = indir->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeIsFloating(indir); + bool needTemp = indir->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeUsesFloatReg(indir); if (addr->AsIntCon()->FitsInAddrBase(m_compiler) && addr->AsIntCon()->AddrNeedsReloc(m_compiler)) { regNumber addrReg = needTemp ? codeGen->internalRegisters.GetSingle(indir) : dataReg; diff --git a/src/coreclr/jit/flowgraph.cpp b/src/coreclr/jit/flowgraph.cpp index b8f1e8d05c999f..842a51c39c7edb 100644 --- a/src/coreclr/jit/flowgraph.cpp +++ b/src/coreclr/jit/flowgraph.cpp @@ -1359,6 +1359,13 @@ bool Compiler::fgCastRequiresHelper(var_types fromType, var_types toType, bool o #endif // TARGET_X86 } #endif // TARGET_X86 || TARGET_ARM +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && (varTypeIsFloating(fromType) || varTypeIsFloating(toType))) + { + // No FP instructions: every conversion to or from floating point is a helper call. + return true; + } +#endif // TARGET_RISCV64 return false; } @@ -2171,7 +2178,7 @@ class MergedReturns retVarDsc->lvType = retLclType; } - if (varTypeIsFloating(retVarDsc->TypeGet())) + if (varTypeIsFloating(retVarDsc->TypeGet()) && varTypeUsesFloatReg(retVarDsc->TypeGet())) { m_compiler->compFloatingPointUsed = true; } diff --git a/src/coreclr/jit/gentree.cpp b/src/coreclr/jit/gentree.cpp index 9b72cec1087afc..dbb95c462526e0 100644 --- a/src/coreclr/jit/gentree.cpp +++ b/src/coreclr/jit/gentree.cpp @@ -34297,11 +34297,14 @@ void ReturnTypeDesc::InitializeStructReturnType(Compiler* comp, m_regType[0] = returnType; #if defined(TARGET_RISCV64) || defined(TARGET_LOONGARCH64) - const CORINFO_FPSTRUCT_LOWERING* lowering = comp->GetFpStructLowering(retClsHnd); - if (!lowering->byIntegerCallConv) + if (!comp->opts.compUseSoftFP) { - assert(lowering->numLoweredElements == 1); - m_fieldOffset[0] = lowering->offsets[0]; + const CORINFO_FPSTRUCT_LOWERING* lowering = comp->GetFpStructLowering(retClsHnd); + if (!lowering->byIntegerCallConv) + { + assert(lowering->numLoweredElements == 1); + m_fieldOffset[0] = lowering->offsets[0]; + } } #endif // defined(TARGET_RISCV64) || defined(TARGET_LOONGARCH64) break; @@ -34373,8 +34376,9 @@ void ReturnTypeDesc::InitializeStructReturnType(Compiler* comp, assert(structSize <= (2 * TARGET_POINTER_SIZE)); BYTE gcPtrs[2] = {TYPE_GC_NONE, TYPE_GC_NONE}; comp->info.compCompHnd->getClassGClayout(retClsHnd, &gcPtrs[0]); - const CORINFO_FPSTRUCT_LOWERING* lowering = comp->GetFpStructLowering(retClsHnd); - if (!lowering->byIntegerCallConv) + const CORINFO_FPSTRUCT_LOWERING* lowering = + comp->opts.compUseSoftFP ? nullptr : comp->GetFpStructLowering(retClsHnd); + if ((lowering != nullptr) && !lowering->byIntegerCallConv) { comp->compFloatingPointUsed = true; assert(lowering->numLoweredElements == MAX_RET_REG_COUNT); diff --git a/src/coreclr/jit/importer.cpp b/src/coreclr/jit/importer.cpp index fe8dd181aab974..89f2fc9309bc41 100644 --- a/src/coreclr/jit/importer.cpp +++ b/src/coreclr/jit/importer.cpp @@ -41,7 +41,7 @@ void Compiler::impPushOnStack(GenTree* tree, typeInfo ti) { compLongUsed = true; } - else if (tree->TypeIs(TYP_FLOAT) || tree->TypeIs(TYP_DOUBLE)) + else if (varTypeIsFloating(tree) && varTypeUsesFloatReg(tree)) { compFloatingPointUsed = true; } diff --git a/src/coreclr/jit/importercalls.cpp b/src/coreclr/jit/importercalls.cpp index cd8c8441aaf4fe..147bccf30c25aa 100644 --- a/src/coreclr/jit/importercalls.cpp +++ b/src/coreclr/jit/importercalls.cpp @@ -5846,7 +5846,8 @@ GenTree* Compiler::impIntrinsic(CORINFO_CLASS_HANDLE clsHnd, #endif // FEATURE_HW_INTRINSICS #ifdef TARGET_RISCV64 - if (!isMagnitude) + // Soft-float: no fmin/fmax, the managed implementation is called instead. + if (!isMagnitude && !opts.compUseSoftFP) { GenTree* op2 = impImplicitR4orR8Cast(impPopStack().val, callType); GenTree* op1 = impImplicitR4orR8Cast(impPopStack().val, callType); @@ -11492,6 +11493,15 @@ GenTree* Compiler::impMathIntrinsic(CORINFO_METHOD_HANDLE method, assert(IsMathIntrinsic(intrinsicName)); assert(isSpecial != nullptr); +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(callType)) + { + // No FP instructions: leave the call to the managed implementation. + JITDUMP("Soft-float: math intrinsic %d is left as a call\n", (int)intrinsicName); + return nullptr; + } +#endif // TARGET_RISCV64 + op1 = nullptr; bool isIntrinsicImplementedByUserCall = IsIntrinsicImplementedByUserCall(intrinsicName); diff --git a/src/coreclr/jit/instr.cpp b/src/coreclr/jit/instr.cpp index cb5c3396233539..9ba41860724fbf 100644 --- a/src/coreclr/jit/instr.cpp +++ b/src/coreclr/jit/instr.cpp @@ -2156,8 +2156,9 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false* else ins = INS_lh; } - else if (TYP_INT == srcType) + else if ((TYP_INT == srcType) || (TYP_FLOAT == srcType)) { + // TYP_FLOAT: soft-float, the value lives in an integer register. ins = INS_lw; } else @@ -2521,8 +2522,8 @@ instruction CodeGenInterface::ins_Store(var_types dstType, bool aligned /*=false ins = INS_sb; else if (varTypeIsShort(dstType)) ins = INS_sh; - else if (TYP_INT == dstType) - ins = INS_sw; + else if ((TYP_INT == dstType) || (TYP_FLOAT == dstType)) + ins = INS_sw; // TYP_FLOAT: soft-float, the value lives in an integer register. else ins = INS_sd; #else diff --git a/src/coreclr/jit/jit.h b/src/coreclr/jit/jit.h index 376fc80e62f661..7fa7a798b5f96d 100644 --- a/src/coreclr/jit/jit.h +++ b/src/coreclr/jit/jit.h @@ -476,6 +476,12 @@ class GlobalJitOptions #else // !FEATURE_HFA static const bool compFeatureHfa = false; #endif // FEATURE_HFA +#ifdef TARGET_RISCV64 + // Soft-float changes the register class of TYP_FLOAT/TYP_DOUBLE (varTypeRegister), + // which is a process-wide table; the mode is fixed by the first compilation + // (see compInitOptions for the states). + static LONG compUseSoftFPConfigured; +#endif // TARGET_RISCV64 #ifdef FEATURE_HFA #undef FEATURE_HFA diff --git a/src/coreclr/jit/jitconfigvalues.h b/src/coreclr/jit/jitconfigvalues.h index fa3a8ee0b99767..5becf2e5d92bc3 100644 --- a/src/coreclr/jit/jitconfigvalues.h +++ b/src/coreclr/jit/jitconfigvalues.h @@ -456,6 +456,7 @@ RELEASE_CONFIG_INTEGER(EnableRiscV64Zba, "EnableRiscV64Zba", RELEASE_CONFIG_INTEGER(EnableRiscV64Zbb, "EnableRiscV64Zbb", 1) // Allows RiscV64 Zbb hardware intrinsics to be disabled RELEASE_CONFIG_INTEGER(EnableRiscV64Zbs, "EnableRiscV64Zbs", 1) // Allows RiscV64 Zbs hardware intrinsics to be disabled RELEASE_CONFIG_INTEGER(EnableRiscV64Zicond, "EnableRiscV64Zicond", 1) // Allows RiscV64 Zicond hardware intrinsics to be disabled +RELEASE_CONFIG_INTEGER(EnableRiscV64Compressed, "EnableRiscV64Compressed", 1) // Allows RiscV64 compressed (C) instruction emission to be disabled #endif RELEASE_CONFIG_INTEGER(EnableEmbeddedBroadcast, "EnableEmbeddedBroadcast", 1) // Allows embedded broadcasts to be disabled diff --git a/src/coreclr/jit/jitee.h b/src/coreclr/jit/jitee.h index 488d012f77babe..231020da41cce7 100644 --- a/src/coreclr/jit/jitee.h +++ b/src/coreclr/jit/jitee.h @@ -40,7 +40,9 @@ class JitFlags #if defined(TARGET_ARM) JIT_FLAG_RELATIVE_CODE_RELOCS = 29, // JIT should generate PC-relative address computations instead of EE relocation records - JIT_FLAG_SOFTFP_ABI = 30, // Enable armel calling convention +#endif +#if defined(TARGET_ARM) || defined(TARGET_RISCV64) + JIT_FLAG_SOFTFP_ABI = 30, // Enable the soft-float calling convention (armel; lp64 on RISC-V) #endif JIT_FLAG_USE_DISPATCH_HELPERS = 31, // The JIT should use helpers for interface dispatch instead of virtual stub dispatch @@ -141,8 +143,10 @@ class JitFlags #if defined(TARGET_ARM) FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_RELATIVE_CODE_RELOCS, JIT_FLAG_RELATIVE_CODE_RELOCS); - FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_SOFTFP_ABI, JIT_FLAG_SOFTFP_ABI); #endif // TARGET_ARM +#if defined(TARGET_ARM) || defined(TARGET_RISCV64) + FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_SOFTFP_ABI, JIT_FLAG_SOFTFP_ABI); +#endif // TARGET_ARM || TARGET_RISCV64 FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_ASYNC, JIT_FLAG_ASYNC); FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_USE_DISPATCH_HELPERS, JIT_FLAG_USE_DISPATCH_HELPERS); FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_VERIFY_GC_MODE_TRANSITIONS, JIT_FLAG_VERIFY_GC_MODE_TRANSITIONS); diff --git a/src/coreclr/jit/lclvars.cpp b/src/coreclr/jit/lclvars.cpp index 0ee7b7042aa4ad..6ba00046a5a3f5 100644 --- a/src/coreclr/jit/lclvars.cpp +++ b/src/coreclr/jit/lclvars.cpp @@ -676,7 +676,10 @@ void Compiler::lvaInitUserArgs(unsigned* curVarNum, unsigned skipArgs, unsigned } #endif - if (info.compIsVarArgs || (opts.compUseSoftFP && varTypeIsFloating(varDsc))) + // Under a soft-float ABI a parameter that lives in a floating-point register + // arrives in an integer one (armel); a parameter whose type is itself in the + // integer register file (RISC-V lp64) is an ordinary integer parameter. + if (info.compIsVarArgs || (opts.compUseSoftFP && varTypeIsFloating(varDsc) && varTypeUsesFloatReg(varDsc))) { #ifndef TARGET_X86 // TODO-CQ: We shouldn't have to go as far as to declare these @@ -897,7 +900,7 @@ void Compiler::lvaInitVarDsc(LclVarDsc* varDsc, } var_types type = JITtype2varType(corInfoType); - if (varTypeIsFloating(type)) + if (varTypeIsFloating(type) && varTypeUsesFloatReg(type)) { compFloatingPointUsed = true; } diff --git a/src/coreclr/jit/lower.cpp b/src/coreclr/jit/lower.cpp index ce5d10adeb9756..649c926f82d018 100644 --- a/src/coreclr/jit/lower.cpp +++ b/src/coreclr/jit/lower.cpp @@ -5478,7 +5478,10 @@ void Lowering::LowerFieldListToFieldListOfRegisters(GenTreeFieldList* fieldLis } // If this is a float -> int insertion, then we need the bitcast now. - if (varTypeUsesFloatReg(value) && varTypeUsesIntReg(regInfo.RegType)) + // (Also checked by type: under a soft-float ABI the FP value already + // lives in an integer register but still needs the integer view for + // the widening and shifting below.) + if ((varTypeUsesFloatReg(value) || varTypeIsFloating(value)) && varTypeUsesIntReg(regInfo.RegType)) { assert((genTypeSize(value) == 4) || (genTypeSize(value) == 8)); var_types castType = genTypeSize(value) == 4 ? TYP_INT : TYP_LONG; diff --git a/src/coreclr/jit/lsra.cpp b/src/coreclr/jit/lsra.cpp index 5803efd8614e12..e024b1d446a36f 100644 --- a/src/coreclr/jit/lsra.cpp +++ b/src/coreclr/jit/lsra.cpp @@ -1009,6 +1009,14 @@ LinearScan::LinearScan(Compiler* theCompiler) availableRegs[static_cast(TYP_##tn)] = ®Fld; #include "typelist.h" #undef DEF_TP +#ifdef TARGET_RISCV64 + if (m_compiler->opts.compUseSoftFP) + { + // No FP register file: float and double values are allocated integer registers. + availableRegs[TYP_FLOAT] = &availableIntRegs; + availableRegs[TYP_DOUBLE] = &availableIntRegs; + } +#endif // TARGET_RISCV64 // Updating lowGprRegs with final value #if defined(TARGET_XARCH) #if defined(TARGET_AMD64) diff --git a/src/coreclr/jit/lsrabuild.cpp b/src/coreclr/jit/lsrabuild.cpp index 099ae3d07221d1..bfb65435071943 100644 --- a/src/coreclr/jit/lsrabuild.cpp +++ b/src/coreclr/jit/lsrabuild.cpp @@ -1105,7 +1105,7 @@ bool LinearScan::buildKillPositionsForNode(GenTree* tree, LsraLocation currentLo } else #endif // FEATURE_PARTIAL_SIMD_CALLEE_SAVE - if (varTypeIsFloating(varDsc) && + if (varTypeIsFloating(varDsc) && varTypeUsesFloatReg(varDsc) && !VarSetOps::IsMember(m_compiler, fpCalleeSaveCandidateVars, varIndex)) { continue; @@ -4330,6 +4330,15 @@ int LinearScan::BuildReturn(GenTree* tree) else #endif // FEATURE_MULTIREG_RET { +#ifdef TARGET_RISCV64 + if (varTypeIsFloating(tree) && !varTypeUsesFloatReg(tree)) + { + // Soft-float: FP values are returned in the integer return register. + BuildUse(op1, RBM_INTRET.GetIntRegSet()); + return 1; + } +#endif // TARGET_RISCV64 + // Non-struct type return - determine useCandidates switch (tree->TypeGet()) { diff --git a/src/coreclr/jit/lsrariscv64.cpp b/src/coreclr/jit/lsrariscv64.cpp index 2531ce69bf4605..5ea3b9cc246ded 100644 --- a/src/coreclr/jit/lsrariscv64.cpp +++ b/src/coreclr/jit/lsrariscv64.cpp @@ -145,7 +145,9 @@ int LinearScan::BuildNode(GenTree* tree) { emitAttr size = emitActualTypeSize(tree); int64_t bits; - if (emitter::isSingleInstructionFpImm(tree->AsDblCon()->DconValue(), size, &bits) && bits != 0) + // Under soft-float the bit pattern is materialized directly into the (integer) target register. + if (!m_compiler->opts.compUseSoftFP && + emitter::isSingleInstructionFpImm(tree->AsDblCon()->DconValue(), size, &bits) && bits != 0) { buildInternalIntRegisterDefForNode(tree); buildInternalRegisterUses(); @@ -561,18 +563,40 @@ int LinearScan::BuildNode(GenTree* tree) GenTree* data = tree->gtGetOp2(); assert(!addr->isContained()); - srcCount = 1; - BuildUse(addr); + // Without the A extension genLockedInstructions expands this to a + // multi-instruction read/modify/write that reuses the address and data + // registers after the first instruction, so their lifetimes have to be + // extended past the def. The arithmetic and bitwise forms also need one + // scratch register for the new value. + const bool plainAtomic = !m_compiler->compOpportunisticallyDependsOn(InstructionSet_A); + + srcCount = 1; + RefPosition* addrUse = BuildUse(addr); + if (plainAtomic) + { + setDelayFree(addrUse); + } if (!data->isContained()) { srcCount++; - BuildUse(data); + RefPosition* dataUse = BuildUse(data); + if (plainAtomic) + { + setDelayFree(dataUse); + } } else { assert(data->IsIntegralConst(0)); } + if (plainAtomic && !tree->OperIs(GT_XCHG)) + { + buildInternalIntRegisterDefForNode(tree); + setInternalRegsDelayFree = true; + buildInternalRegisterUses(); + } + if (dstCount == 1) { BuildDef(tree); @@ -858,7 +882,7 @@ int LinearScan::BuildIndir(GenTreeIndir* indirTree) addr->AsIntCon()->FitsInAddrBase(m_compiler) && addr->AsIntCon()->AddrNeedsReloc(m_compiler); if (needsReloc || !emitter::isValidSimm12(indirTree->Offset())) { - bool needTemp = indirTree->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeIsFloating(indirTree); + bool needTemp = indirTree->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeUsesFloatReg(indirTree); if (needTemp) { // This offset can't be contained in the ld/sd instruction, so we need an internal register diff --git a/src/coreclr/jit/morph.cpp b/src/coreclr/jit/morph.cpp index 28092b740b13fa..f5d3d34cee3668 100644 --- a/src/coreclr/jit/morph.cpp +++ b/src/coreclr/jit/morph.cpp @@ -253,6 +253,337 @@ GenTree* Compiler::fgMorphIntoHelperCall(GenTree* tree, int helper, bool morphAr return tree; } +#ifdef TARGET_RISCV64 + +//------------------------------------------------------------------------ +// Soft-float (RISC-V without the F/D extensions) +// +// Under opts.compUseSoftFP there are no floating-point instructions or +// registers: float/double values live in integer registers (see +// Compiler::compInitOptions), and every floating-point arithmetic operation, +// comparison and conversion is expanded during global morph into a call to +// one of the soft-float helpers, so that no such node survives into the +// backend. The IR keeps its floating-point types, so value numbering, +// constant folding and CSE work unchanged; the helpers are modelled as the +// operations they implement (see fgValueNumberJitHelperMethodVNFunc). +// + +//------------------------------------------------------------------------ +// fgMorphSoftFloatArith: expand an FP arithmetic operation into a helper call +// +// Arguments: +// tree - the GT_ADD, GT_SUB, GT_MUL or GT_DIV node +// +// Return Value: +// The morphed replacement tree. +// +// Notes: +// Unlike USE_HELPER_FOR_ARITH this builds a new call node: the importer +// only allocates add/sub as small nodes, which cannot be turned into a +// call in place. +// +GenTree* Compiler::fgMorphSoftFloatArith(GenTreeOp* tree) +{ + assert(opts.compUseSoftFP && tree->OperIs(GT_ADD, GT_SUB, GT_MUL, GT_DIV) && varTypeIsFloating(tree)); + + const var_types type = tree->TypeGet(); + + // The IL stack has a single F type, so the operands may differ in size + // from the result; the helpers take operands of the result type. + for (GenTree** use : {&tree->gtOp1, &tree->gtOp2}) + { + if ((*use)->TypeGet() != type) + { + *use = gtNewCastNode(type, *use, false, type); + } + } + + GenTree* folded = gtFoldExpr(tree); + if (folded != tree) + { + return fgMorphTree(folded); + } + if (folded->OperIsLeaf()) + { + return fgMorphLeaf(folded); + } + + const bool isFloat = (type == TYP_FLOAT); + CorInfoHelpFunc helper; + switch (tree->OperGet()) + { + case GT_ADD: + helper = isFloat ? CORINFO_HELP_FLTADD : CORINFO_HELP_DBLADD; + break; + case GT_SUB: + helper = isFloat ? CORINFO_HELP_FLTSUB : CORINFO_HELP_DBLSUB; + break; + case GT_MUL: + helper = isFloat ? CORINFO_HELP_FLTMUL : CORINFO_HELP_DBLMUL; + break; + default: + helper = isFloat ? CORINFO_HELP_FLTDIV : CORINFO_HELP_DBLDIV; + break; + } + + GenTreeCall* call = gtNewHelperCallNode(helper, type, tree->gtGetOp1(), tree->gtGetOp2()); + return fgMorphTree(call); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatCast: expand a conversion into a helper call +// +// Arguments: +// tree - the cast node +// helper - the conversion helper +// oper - the (possibly widened) operand, tree's cast operand +// +// Return Value: +// The morphed replacement tree. +// +// Notes: +// The counterpart of fgMorphCastIntoHelper that builds a new node: casts +// created after import (impImplicitR4orR8Cast, the widening in +// fgMorphExpandCast) are not large enough to become a call in place. +// +GenTree* Compiler::fgMorphSoftFloatCast(GenTreeCast* tree, CorInfoHelpFunc helper, GenTree* oper) +{ + assert(opts.compUseSoftFP && (tree->CastOp() == oper)); + + if (oper->OperIsConst()) + { + GenTree* folded = gtFoldExprConst(tree); + if (folded != tree) + { + return fgMorphTree(folded); + } + if (folded->OperIsConst()) + { + return fgMorphConst(folded); + } + } + + GenTreeCall* call = gtNewHelperCallNode(helper, genActualType(tree->CastToType()), oper); + return fgMorphTree(call); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatNeg: expand an FP negation into a sign-bit flip +// +// Arguments: +// neg - the GT_NEG node +// +// Return Value: +// The replacement tree, not yet morphed. +// +GenTree* Compiler::fgMorphSoftFloatNeg(GenTreeOp* neg) +{ + assert(opts.compUseSoftFP && neg->OperIs(GT_NEG) && varTypeIsFloating(neg)); + + const var_types type = neg->TypeGet(); + GenTree* op = neg->gtGetOp1(); + + if (op->IsCnsFltOrDbl()) + { + return gtNewDconNode(-op->AsDblCon()->DconValue(), type); + } + + // -x == x ^ signBit, computed in the integer domain. + const var_types intType = (type == TYP_FLOAT) ? TYP_INT : TYP_LONG; + GenTree* signBit = (type == TYP_FLOAT) ? gtNewIconNode(INT32_MIN, TYP_INT) : gtNewLconNode(INT64_MIN); + GenTree* bits = gtNewOperNode(GT_XOR, intType, gtNewBitCastNode(intType, op), signBit); + return gtNewBitCastNode(type, bits); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatRelop: expand an FP comparison into a compare helper call +// +// Arguments: +// relop - the comparison node +// +// Return Value: +// The replacement tree, an integer comparison of the helper result, not yet morphed. +// +// Notes: +// The helpers return a three-way result and differ only for unordered +// operands (CMP_LE returns 1, CMP_GE returns -1), which lets each IL form +// be expressed with a single call and a comparison of its result with 0: +// +// oeq: LE == 0 une: LE != 0 +// olt: LE < 0 ult: GE < 0 +// ole: LE <= 0 ule: GE <= 0 +// ogt: GE > 0 ugt: LE > 0 +// oge: GE >= 0 uge: LE >= 0 +// +// ueq and one need both results: +// +// ueq = (LE >= 0) && (GE <= 0) one = (LE < 0) || (GE > 0) +// +GenTree* Compiler::fgMorphSoftFloatRelop(GenTreeOp* relop) +{ + assert(opts.compUseSoftFP && relop->OperIsCompare()); + + GenTree* op1 = relop->gtGetOp1(); + GenTree* op2 = relop->gtGetOp2(); + assert(varTypeIsFloating(op1) && (op1->TypeGet() == op2->TypeGet())); + + const bool isFloat = op1->TypeIs(TYP_FLOAT); + const bool isUnordered = (relop->gtFlags & GTF_RELOP_NAN_UN) != 0; + const genTreeOps oper = relop->OperGet(); + const CorInfoHelpFunc cmpLE = isFloat ? CORINFO_HELP_FLTCMP_LE : CORINFO_HELP_DBLCMP_LE; + const CorInfoHelpFunc cmpGE = isFloat ? CORINFO_HELP_FLTCMP_GE : CORINFO_HELP_DBLCMP_GE; + + GenTree* result; + if (oper == (isUnordered ? GT_EQ : GT_NE)) + { + // ueq / one: both helpers are needed, so the operands go into temps. + // The stores are placed in the first operand of the AND/OR; the two + // operands both contain calls, so their evaluation order is fixed. + TempInfo tmp1 = fgMakeTemp(op1); + TempInfo tmp2 = fgMakeTemp(op2); + + GenTree* le = gtNewHelperCallNode(cmpLE, TYP_INT, tmp1.load, tmp2.load); + le = gtNewOperNode(GT_COMMA, TYP_INT, tmp1.store, gtNewOperNode(GT_COMMA, TYP_INT, tmp2.store, le)); + GenTree* ge = gtNewHelperCallNode(cmpGE, TYP_INT, gtCloneExpr(tmp1.load), gtCloneExpr(tmp2.load)); + + GenTree* cmp; + if (oper == GT_EQ) + { + cmp = gtNewOperNode(GT_AND, TYP_INT, gtNewOperNode(GT_GE, TYP_INT, le, gtNewIconNode(0)), + gtNewOperNode(GT_LE, TYP_INT, ge, gtNewIconNode(0))); + } + else + { + cmp = gtNewOperNode(GT_OR, TYP_INT, gtNewOperNode(GT_LT, TYP_INT, le, gtNewIconNode(0)), + gtNewOperNode(GT_GT, TYP_INT, ge, gtNewIconNode(0))); + } + // Keep the tree rooted at a comparison, for GT_JTRUE users. + result = gtNewOperNode(GT_NE, TYP_INT, cmp, gtNewIconNode(0)); + } + else + { + CorInfoHelpFunc helper; + switch (oper) + { + case GT_EQ: + case GT_NE: + helper = cmpLE; + break; + case GT_LT: + case GT_LE: + helper = isUnordered ? cmpGE : cmpLE; + break; + default: + assert((oper == GT_GT) || (oper == GT_GE)); + helper = isUnordered ? cmpLE : cmpGE; + break; + } + GenTree* call = gtNewHelperCallNode(helper, TYP_INT, op1, op2); + result = gtNewOperNode(oper, TYP_INT, call, gtNewIconNode(0)); + } + + // A JTRUE/QMARK condition carries GTF_RELOP_JMP_USED and GTF_DONT_CSE (set by + // the parent before its operands are morphed); the replacement must keep both + // so that CSE does not turn the condition into a local. + result->gtFlags |= (relop->gtFlags & (GTF_RELOP_JMP_USED | GTF_DONT_CSE)); + return result; +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatCkFinite: expand GT_CKFINITE into an exponent check +// +// Arguments: +// ckFinite - the GT_CKFINITE node +// +// Return Value: +// The replacement tree, not yet morphed: +// +// COMMA(tmp = x, COMMA(BOUNDS_CHECK((bits(tmp) >> expShift) & expMask, expMask), tmp)) +// +// NaN and the infinities have all exponent bits set, so the bounds check +// (index >= length) throws exactly for them. +// +GenTree* Compiler::fgMorphSoftFloatCkFinite(GenTreeOp* ckFinite) +{ + assert(opts.compUseSoftFP && ckFinite->OperIs(GT_CKFINITE)); + + const var_types type = ckFinite->TypeGet(); + const bool isFloat = (type == TYP_FLOAT); + const var_types intType = isFloat ? TYP_INT : TYP_LONG; + const int expBits = isFloat ? 8 : 11; + const int expMask = (1 << expBits) - 1; + + TempInfo tmp = fgMakeTemp(ckFinite->gtGetOp1()); + GenTree* bits = gtNewBitCastNode(intType, tmp.load); + GenTree* exp = gtNewOperNode(GT_RSZ, intType, bits, gtNewIconNode(genTypeSize(type) * 8 - 1 - expBits)); + exp = gtNewOperNode(GT_AND, intType, exp, isFloat ? gtNewIconNode(expMask) : gtNewLconNode(expMask)); + if (!isFloat) + { + exp = gtNewCastNode(TYP_INT, exp, false, TYP_INT); + } + GenTree* check = new (this, GT_BOUNDS_CHECK) GenTreeBoundsChk(exp, gtNewIconNode(expMask), SCK_ARITH_EXCPN); + + GenTree* result = gtNewOperNode(GT_COMMA, type, check, gtCloneExpr(tmp.load)); + return gtNewOperNode(GT_COMMA, type, tmp.store, result); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatCastToInt32: expand a non-overflow double -> int/uint cast +// +// Arguments: +// src - the TYP_DOUBLE source +// toUnsigned - true for uint +// +// Return Value: +// The replacement tree, not yet morphed. +// +// Notes: +// The 64-bit conversion helpers implement the .NET semantics (NaN -> 0, +// saturation); their result is saturated to the 32-bit range here. The +// clamping is branchless and does not use GT_SELECT, which on RISC-V needs +// Zicond. The temps are stored first in an explicit COMMA chain since the +// store-before-use dependency is not otherwise expressed in the IR. +// +GenTree* Compiler::fgMorphSoftFloatCastToInt32(GenTree* src, bool toUnsigned) +{ + assert(opts.compUseSoftFP && src->TypeIs(TYP_DOUBLE)); + + GenTreeCall* cvt = gtNewHelperCallNode(toUnsigned ? CORINFO_HELP_DBL2ULNG : CORINFO_HELP_DBL2LNG, TYP_LONG, src); + TempInfo tmpVal = fgMakeTemp(cvt); + + if (toUnsigned) + { + // result = (uint)val | -(val > UINT32_MAX); the helper never returns a negative value. + GenTree* isHi = gtNewOperNode(GT_GT, TYP_INT, gtCloneExpr(tmpVal.load), gtNewLconNode((int64_t)UINT32_MAX)); + isHi->gtFlags |= GTF_UNSIGNED; + GenTree* trunc = gtNewCastNode(TYP_INT, tmpVal.load, false, TYP_UINT); + GenTree* result = gtNewOperNode(GT_OR, TYP_INT, trunc, gtNewOperNode(GT_NEG, TYP_INT, isHi)); + return gtNewOperNode(GT_COMMA, TYP_INT, tmpVal.store, result); + } + + // maskHi = -(val > INT32_MAX); maskLo = -(val < INT32_MIN); + // result = ((int)val & ~(maskHi | maskLo)) | (INT32_MAX & maskHi) | (INT32_MIN & maskLo) + TempInfo tmpHi = + fgMakeTemp(gtNewOperNode(GT_NEG, TYP_INT, + gtNewOperNode(GT_GT, TYP_INT, gtCloneExpr(tmpVal.load), gtNewLconNode(INT32_MAX)))); + TempInfo tmpLo = + fgMakeTemp(gtNewOperNode(GT_NEG, TYP_INT, + gtNewOperNode(GT_LT, TYP_INT, gtCloneExpr(tmpVal.load), gtNewLconNode(INT32_MIN)))); + + GenTree* trunc = gtNewCastNode(TYP_INT, tmpVal.load, false, TYP_INT); + GenTree* outMask = gtNewOperNode(GT_OR, TYP_INT, tmpHi.load, tmpLo.load); + GenTree* inVal = gtNewOperNode(GT_AND, TYP_INT, trunc, gtNewOperNode(GT_NOT, TYP_INT, outMask)); + GenTree* hiVal = gtNewOperNode(GT_AND, TYP_INT, gtNewIconNode(INT32_MAX, TYP_INT), gtCloneExpr(tmpHi.load)); + GenTree* loVal = gtNewOperNode(GT_AND, TYP_INT, gtNewIconNode(INT32_MIN, TYP_INT), gtCloneExpr(tmpLo.load)); + GenTree* result = gtNewOperNode(GT_OR, TYP_INT, gtNewOperNode(GT_OR, TYP_INT, inVal, hiVal), loVal); + + result = gtNewOperNode(GT_COMMA, TYP_INT, tmpLo.store, result); + result = gtNewOperNode(GT_COMMA, TYP_INT, tmpHi.store, result); + return gtNewOperNode(GT_COMMA, TYP_INT, tmpVal.store, result); +} + +#endif // TARGET_RISCV64 + //------------------------------------------------------------------------ // fgMorphExpandCast: Performs the pre-order (required) morphing for a cast. // @@ -282,6 +613,7 @@ GenTree* Compiler::fgMorphIntoHelperCall(GenTree* tree, int helper, bool morphAr // in which case the cast may be transformed into an unchecked one // and its operand changed (the cast "expanded" into two). // + GenTree* Compiler::fgMorphExpandCast(GenTreeCast* tree) { GenTree* oper = tree->CastOp(); @@ -445,6 +777,23 @@ GenTree* Compiler::fgMorphExpandCast(GenTreeCast* tree) case TYP_ULONG: helper = CORINFO_HELP_DBL2ULNG; break; +#ifdef TARGET_RISCV64 + case TYP_INT: + case TYP_UINT: + { + // Soft-float: convert with the 64-bit helper and saturate to 32 bits. + assert(opts.compUseSoftFP); + if (tree->CastOp()->OperIsConst()) + { + GenTree* folded = gtFoldExprConst(tree); + if (folded != tree) + { + return fgMorphTree(folded); + } + } + return fgMorphTree(fgMorphSoftFloatCastToInt32(oper, dstType == TYP_UINT)); + } +#endif // TARGET_RISCV64 default: unreached(); } @@ -469,6 +818,42 @@ GenTree* Compiler::fgMorphExpandCast(GenTreeCast* tree) return fgMorphTree(oper); } +#ifdef TARGET_RISCV64 + else if (opts.compUseSoftFP && varTypeIsFloating(dstType)) + { + // Soft-float: conversions to floating point are helper calls. + if (varTypeIsFloating(srcType)) + { + if (srcType == dstType) + { + return fgMorphTree(oper); + } + return fgMorphSoftFloatCast(tree, (dstType == TYP_DOUBLE) ? CORINFO_HELP_FLT2DBL : CORINFO_HELP_DBL2FLT, + oper); + } + + assert(varTypeIsIntegral(srcType)); + if (!varTypeIsLong(srcType)) + { + // Widen to 64 bits first; a zero-extended uint is exact in the signed helper. + oper = gtNewCastNode(TYP_LONG, oper, tree->IsUnsigned(), TYP_LONG); + tree->ClearUnsigned(); + tree->CastOp() = oper; + } + + CorInfoHelpFunc helper; + if (dstType == TYP_FLOAT) + { + helper = tree->IsUnsigned() ? CORINFO_HELP_ULNG2FLT : CORINFO_HELP_LNG2FLT; + } + else + { + helper = tree->IsUnsigned() ? CORINFO_HELP_ULNG2DBL : CORINFO_HELP_LNG2DBL; + } + return fgMorphSoftFloatCast(tree, helper, oper); + } +#endif // TARGET_RISCV64 + #ifndef TARGET_64BIT else if (varTypeIsLong(srcType)) { @@ -7011,6 +7396,40 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) // Some arithmetic operators need to use a helper call to the EE int helper; +#ifdef TARGET_RISCV64 + // Soft-float: expand the FP operations into helper calls before morphing their operands. + case GT_ADD: + case GT_SUB: + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphSoftFloatArith(tree->AsOp()); + } + break; + + case GT_NEG: + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphTree(fgMorphSoftFloatNeg(tree->AsOp())); + } + break; + + case GT_LT: + case GT_LE: + case GT_GE: + if (opts.compUseSoftFP && varTypeIsFloating(op1)) + { + return fgMorphTree(fgMorphSoftFloatRelop(tree->AsOp())); + } + break; + + case GT_CKFINITE: + if (opts.compUseSoftFP) + { + return fgMorphTree(fgMorphSoftFloatCkFinite(tree->AsOp())); + } + break; +#endif // TARGET_RISCV64 + case GT_STORE_LCL_VAR: case GT_STORE_LCL_FLD: { @@ -7085,6 +7504,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) case GT_MUL: noway_assert(op2 != nullptr); +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphSoftFloatArith(tree->AsOp()); + } +#endif // TARGET_RISCV64 + #if !defined(TARGET_64BIT) && !defined(TARGET_WASM) if (typ == TYP_LONG) { @@ -7189,6 +7615,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) break; case GT_DIV: +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphSoftFloatArith(tree->AsOp()); + } +#endif // TARGET_RISCV64 + // Convert DIV to UDIV if both op1 and op2 are known to be never negative if (varTypeIsIntegral(tree) && op1->IsNeverNegative(this) && op2->IsNeverNegative(this)) { @@ -7484,6 +7917,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) case GT_EQ: case GT_NE: { +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(op1)) + { + return fgMorphTree(fgMorphSoftFloatRelop(tree->AsOp())); + } +#endif // TARGET_RISCV64 + if (opts.OptimizationEnabled()) { GenTree* optimizedTree = gtFoldTypeCompare(tree); @@ -7545,6 +7985,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) case GT_GT: { +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(op1)) + { + return fgMorphTree(fgMorphSoftFloatRelop(tree->AsOp())); + } +#endif // TARGET_RISCV64 + // Try and optimize nullable boxes feeding compares GenTree* optimizedTree = gtFoldBoxNullable(tree); diff --git a/src/coreclr/jit/targetriscv64.cpp b/src/coreclr/jit/targetriscv64.cpp index 5ba2a3100b90e8..b60807496244f3 100644 --- a/src/coreclr/jit/targetriscv64.cpp +++ b/src/coreclr/jit/targetriscv64.cpp @@ -78,7 +78,7 @@ ABIPassingInformation RiscV64Classifier::Classify(Compiler* comp, passedByRef = true; passedSize = TARGET_POINTER_SIZE; } - else if (!structLayout->IsBlockLayout()) + else if (!structLayout->IsBlockLayout() && !comp->opts.compUseSoftFP) { lowering = comp->GetFpStructLowering(structLayout->GetClassHandle()); if (!lowering->byIntegerCallConv) @@ -100,7 +100,7 @@ ABIPassingInformation RiscV64Classifier::Classify(Compiler* comp, { passedSize = genTypeSize(type); assert(passedSize <= TARGET_POINTER_SIZE); - floatFields = varTypeIsFloating(type) ? 1 : 0; + floatFields = (varTypeIsFloating(type) && !comp->opts.compUseSoftFP) ? 1 : 0; } assert((floatFields > 0) || (intFields == 0)); diff --git a/src/coreclr/jit/utils.cpp b/src/coreclr/jit/utils.cpp index 17263b34b48ec5..f2724acf343dce 100644 --- a/src/coreclr/jit/utils.cpp +++ b/src/coreclr/jit/utils.cpp @@ -86,7 +86,11 @@ const BYTE varTypeClassification[] = { #undef DEF_TP }; +#ifdef TARGET_RISCV64 +BYTE varTypeRegister[] = { +#else const BYTE varTypeRegister[] = { +#endif #define DEF_TP(tn, nm, jitType, sz, sze, asze, st, al, regTyp, regFld, csr, ctr, tf) regTyp, #include "typelist.h" #undef DEF_TP @@ -1458,6 +1462,21 @@ void HelperCallProperties::init() case CORINFO_HELP_LLSH: case CORINFO_HELP_LRSH: case CORINFO_HELP_LRSZ: + // Soft-float helpers: leaf compiler-rt routines, no GC interaction. + case CORINFO_HELP_FLTADD: + case CORINFO_HELP_FLTSUB: + case CORINFO_HELP_FLTMUL: + case CORINFO_HELP_FLTDIV: + case CORINFO_HELP_DBLADD: + case CORINFO_HELP_DBLSUB: + case CORINFO_HELP_DBLMUL: + case CORINFO_HELP_DBLDIV: + case CORINFO_HELP_FLTCMP_LE: + case CORINFO_HELP_FLTCMP_GE: + case CORINFO_HELP_DBLCMP_LE: + case CORINFO_HELP_DBLCMP_GE: + case CORINFO_HELP_FLT2DBL: + case CORINFO_HELP_DBL2FLT: isNoGC = true; FALLTHROUGH; case CORINFO_HELP_LMUL: diff --git a/src/coreclr/jit/valuenum.cpp b/src/coreclr/jit/valuenum.cpp index 95ab658054d937..88fcac3c02e44d 100644 --- a/src/coreclr/jit/valuenum.cpp +++ b/src/coreclr/jit/valuenum.cpp @@ -15206,6 +15206,14 @@ void Compiler::fgValueNumberCastHelper(GenTreeCall* call) castFromType = TYP_DOUBLE; hasOverflowCheck = true; break; + case CORINFO_HELP_FLT2DBL: + castToType = TYP_DOUBLE; + castFromType = TYP_FLOAT; + break; + case CORINFO_HELP_DBL2FLT: + castToType = TYP_FLOAT; + castFromType = TYP_DOUBLE; + break; default: unreached(); @@ -15273,6 +15281,42 @@ VNFunc Compiler::fgValueNumberJitHelperMethodVNFunc(CorInfoHelpFunc helpFunc) case CORINFO_HELP_DBLREM: vnf = VNF_MOD; break; + case CORINFO_HELP_FLTADD: + vnf = VNFunc(GT_ADD); + break; + case CORINFO_HELP_DBLADD: + vnf = VNFunc(GT_ADD); + break; + case CORINFO_HELP_FLTSUB: + vnf = VNFunc(GT_SUB); + break; + case CORINFO_HELP_DBLSUB: + vnf = VNFunc(GT_SUB); + break; + case CORINFO_HELP_FLTMUL: + vnf = VNFunc(GT_MUL); + break; + case CORINFO_HELP_DBLMUL: + vnf = VNFunc(GT_MUL); + break; + case CORINFO_HELP_FLTDIV: + vnf = VNFunc(GT_DIV); + break; + case CORINFO_HELP_DBLDIV: + vnf = VNFunc(GT_DIV); + break; + case CORINFO_HELP_FLTCMP_LE: + vnf = VNF_SoftFPCmpLE; + break; + case CORINFO_HELP_DBLCMP_LE: + vnf = VNF_SoftFPCmpLE; + break; + case CORINFO_HELP_FLTCMP_GE: + vnf = VNF_SoftFPCmpGE; + break; + case CORINFO_HELP_DBLCMP_GE: + vnf = VNF_SoftFPCmpGE; + break; // These allocation operations probably require some augmentation -- perhaps allocSiteId, // something about array length... @@ -15515,6 +15559,8 @@ bool Compiler::fgValueNumberHelperCall(GenTreeCall* call) case CORINFO_HELP_DBL2LNG: case CORINFO_HELP_DBL2LNG_OVF: case CORINFO_HELP_DBL2UINT_OVF: + case CORINFO_HELP_FLT2DBL: + case CORINFO_HELP_DBL2FLT: case CORINFO_HELP_DBL2ULNG: case CORINFO_HELP_DBL2ULNG_OVF: fgValueNumberCastHelper(call); diff --git a/src/coreclr/jit/valuenumfuncs.h b/src/coreclr/jit/valuenumfuncs.h index f063957255e964..647103a06ea2dc 100644 --- a/src/coreclr/jit/valuenumfuncs.h +++ b/src/coreclr/jit/valuenumfuncs.h @@ -208,6 +208,12 @@ ValueNumFuncDef(HWI_INTRINSIC_END, -1, false, false) #define VNF_HWI_LAST (VNF_HWI_INTRINSIC_END - 1) #endif // FEATURE_HW_INTRINSICS + // Three-way floating-point comparisons of the soft-float compare helpers + // (CORINFO_HELP_FLTCMP_LE & co). Not commutative: swapping the operands + // negates the result, and the two forms differ for unordered operands. + ValueNumFuncDef(SoftFPCmpLE, 2, false, false) + ValueNumFuncDef(SoftFPCmpGE, 2, false, false) + #if defined(TARGET_RISCV64) // Signed/Unsigned integer min/max intrinsics ValueNumFuncDef(MinInt, 2, true, false) diff --git a/src/coreclr/jit/vartype.h b/src/coreclr/jit/vartype.h index c501dea656ebda..e89f7e5518e8bd 100644 --- a/src/coreclr/jit/vartype.h +++ b/src/coreclr/jit/vartype.h @@ -48,7 +48,13 @@ enum var_types_register /*****************************************************************************/ const extern BYTE varTypeClassification[TYP_COUNT]; +#ifdef TARGET_RISCV64 +// Not const: a soft-float (no F/D) process moves TYP_FLOAT/TYP_DOUBLE to the +// integer register file, see Compiler::compInitOptions. +extern BYTE varTypeRegister[TYP_COUNT]; +#else const extern BYTE varTypeRegister[TYP_COUNT]; +#endif // make any class with a TypeGet member also have a function TypeGet() that does the same thing template @@ -339,6 +345,10 @@ inline bool varTypeUsesFloatArgReg(T vt) // Exception: Windows arm64 native varargs passes them using general-purpose (integer) registers or // by value on the stack, or split between registers and stack. return varTypeUsesFloatReg(vt); +#elif defined(TARGET_RISCV64) + // The register class of TYP_FLOAT/TYP_DOUBLE follows the ABI: under the soft-float + // (lp64) ABI they live in, and are returned in, integer registers. + return varTypeUsesFloatReg(vt); #else // Other targets pass them as regular structs - by reference or by value. return varTypeIsFloating(vt); diff --git a/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets b/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets index 368ab3b7b50c3d..be8043b8ee3089 100644 --- a/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets +++ b/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets @@ -44,6 +44,12 @@ <_targetArchitectureWithAbi>$(_targetArchitecture) <_targetArchitectureWithAbi Condition="'$(_linuxLibcFlavor)' == 'bionic' and '$(_targetArchitecture)' == 'arm'">armel + + + <_targetArchitectureWithAbi Condition="'$(IlcRiscV64SoftFloat)' == 'true' and '$(_targetArchitecture)' == 'riscv64'">riscv64-lp64 diff --git a/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets b/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets index 5b6862703de826..fb685cced1409a 100644 --- a/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets +++ b/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets @@ -261,6 +261,12 @@ The .NET Foundation licenses this file to you under the MIT license. + + diff --git a/src/coreclr/nativeaot/Runtime/MathHelpers.cpp b/src/coreclr/nativeaot/Runtime/MathHelpers.cpp index 2cd1075b047502..f33d13ada2031b 100644 --- a/src/coreclr/nativeaot/Runtime/MathHelpers.cpp +++ b/src/coreclr/nativeaot/Runtime/MathHelpers.cpp @@ -9,26 +9,22 @@ // Floating point and 64-bit integer math helpers. // +// The .NET semantics of the floating point to integer conversions (NaN -> 0, +// saturation) are spelled out rather than left to the C cast: the cast only +// happens to have them where the hardware instruction does, and a soft-float +// build lowers it to a compiler-rt routine that has neither. FCIMPL1_D(uint64_t, RhpDbl2ULng, double val) { -#if defined(HOST_X86) || defined(HOST_AMD64) const double uint64_max_plus_1 = 4294967296.0 * 4294967296.0; return (val > 0) ? ((val >= uint64_max_plus_1) ? UINT64_MAX : (uint64_t)val) : 0; -#else - return (uint64_t)val; -#endif } FCIMPLEND FCIMPL1_D(int64_t, RhpDbl2Lng, double val) { -#if defined(HOST_X86) || defined(HOST_AMD64) || defined(HOST_ARM) const double int64_min = -2147483648.0 * 4294967296.0; const double int64_max = 2147483648.0 * 4294967296.0; return (val != val) ? 0 : (val <= int64_min) ? INT64_MIN : (val >= int64_max) ? INT64_MAX : (int64_t)val; -#else - return (int64_t)val; -#endif } FCIMPLEND @@ -61,6 +57,12 @@ FCIMPL2_LL(uint64_t, ModUInt64Internal, uint64_t i, uint64_t j) } FCIMPLEND +#endif + +// The int64 -> floating point conversion helpers are used where the JIT cannot +// emit the conversion inline: 32-bit targets and the soft-float RISC-V ABI. +#if !defined(HOST_64BIT) || defined(HOST_RISCV64) + FCIMPL1_L(double, RhpLng2Dbl, int64_t val) { return (double)val; diff --git a/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S b/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S index 8cbe1b6a276982..21b1ac1b2e22a7 100644 --- a/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S +++ b/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S @@ -31,6 +31,7 @@ .endif // Safely using available registers for floating-point saves +#if __riscv_flen >= 64 fsd fs0, 0x10(sp) fsd fs1, 0x18(sp) fsd fs2, 0x20(sp) @@ -43,6 +44,7 @@ fsd fs9, 0x58(sp) fsd fs10, 0x60(sp) fsd fs11, 0x68(sp) +#endif PROLOG_SAVE_REG_PAIR_INDEXED fp, ra, 0x78 @@ -140,6 +142,7 @@ // Load FP preserved registers // addi t3, \regdisplayReg, OFFSETOF__REGDISPLAY__F // Base address of floating-point registers +#if __riscv_flen >= 64 fld fs0, 0x40(t3) // Load fs0 fld fs1, 0x48(t3) // Load fs1 fld fs2, 0x90(t3) // Load fs2 @@ -152,6 +155,7 @@ fld fs9, 0xc8(t3) // Load fs9 fld fs10, 0xd0(t3) // Load fs10 fld fs11, 0xd8(t3) // Load fs11 +#endif .endm @@ -188,6 +192,7 @@ // Save floating-point registers addi t3, \regdisplayReg, OFFSETOF__REGDISPLAY__F +#if __riscv_flen >= 64 fsd fs0, 0x40(t3) fsd fs1, 0x48(t3) fsd fs2, 0x90(t3) @@ -200,6 +205,7 @@ fsd fs9, 0xc8(t3) fsd fs10, 0xd0(t3) fsd fs11, 0xd8(t3) +#endif .endm @@ -474,6 +480,7 @@ LOCAL_LABEL(NotHijacked): ALLOC_CALL_FUNCLET_FRAME 0x90 // Save floating-point registers +#if __riscv_flen >= 64 fsd fs0, 0x00(sp) fsd fs1, 0x08(sp) fsd fs2, 0x10(sp) @@ -486,6 +493,7 @@ LOCAL_LABEL(NotHijacked): fsd fs9, 0x48(sp) fsd fs10, 0x50(sp) fsd fs11, 0x58(sp) +#endif // Save integer registers sd a0, 0x60(sp) // Save a0 to a3 @@ -511,7 +519,14 @@ LOCAL_LABEL(NotHijacked): addi t3, a5, OFFSETOF__Thread__m_ThreadStateFlags addiw a6, zero, -17 // Mask value (0xFFFFFFEF) +#ifdef __riscv_atomic amoand.w a4, a6, (t3) +#else + // No A extension: non-atomic read-modify-write. + lw a4, (t3) + and a4, a4, a6 + sw a4, (t3) +#endif // Set preserved regs to the values expected by the funclet RESTORE_PRESERVED_REGISTERS a2 @@ -593,6 +608,7 @@ LOCAL_LABEL(DonePopping): ALLOC_CALL_FUNCLET_FRAME 0x80 // Save floating-point registers +#if __riscv_flen >= 64 fsd fs0, 0x00(sp) fsd fs1, 0x08(sp) fsd fs2, 0x10(sp) @@ -605,6 +621,7 @@ LOCAL_LABEL(DonePopping): fsd fs9, 0x48(sp) fsd fs10, 0x50(sp) fsd fs11, 0x58(sp) +#endif // Save integer registers sd a0, 0x60(sp) // Save a0 to 0x60 @@ -623,7 +640,14 @@ LOCAL_LABEL(DonePopping): // Set the DoNotTriggerGc flag addi t3, a2, OFFSETOF__Thread__m_ThreadStateFlags addiw a3, zero, -17 // Mask value (0xFFFFFFEF) +#ifdef __riscv_atomic amoand.w a4, a3, (t3) +#else + // No A extension: non-atomic read-modify-write. + lw a4, (t3) + and a4, a4, a3 + sw a4, (t3) +#endif // Restore preserved registers RESTORE_PRESERVED_REGISTERS a1 @@ -646,9 +670,17 @@ LOCAL_LABEL(DonePopping): addi t3, a2, OFFSETOF__Thread__m_ThreadStateFlags addiw a3, zero, 16 // Mask value (0x10) +#ifdef __riscv_atomic amoor.w a1, a3, (t3) +#else + // No A extension: non-atomic read-modify-write. + lw a1, (t3) + or a1, a1, a3 + sw a1, (t3) +#endif // Restore floating-point registers +#if __riscv_flen >= 64 fld fs0, 0x00(sp) fld fs1, 0x08(sp) fld fs2, 0x10(sp) @@ -661,6 +693,7 @@ LOCAL_LABEL(DonePopping): fld fs9, 0x48(sp) fld fs10, 0x50(sp) fld fs11, 0x58(sp) +#endif // Free call funclet frame FREE_CALL_FUNCLET_FRAME 0x80 @@ -685,6 +718,7 @@ LOCAL_LABEL(DonePopping): NESTED_ENTRY RhpCallFilterFunclet, _TEXT, NoHandler ALLOC_CALL_FUNCLET_FRAME 0x60 +#if __riscv_flen >= 64 fsd fs0, 0x00(sp) fsd fs1, 0x08(sp) fsd fs2, 0x10(sp) @@ -697,6 +731,7 @@ LOCAL_LABEL(DonePopping): fsd fs9, 0x48(sp) fsd fs10, 0x50(sp) fsd fs11, 0x58(sp) +#endif ld t3, OFFSETOF__REGDISPLAY__pFP(a2) ld fp, 0(t3) @@ -709,6 +744,7 @@ LOCAL_LABEL(DonePopping): ALTERNATE_ENTRY RhpCallFilterFunclet2 +#if __riscv_flen >= 64 fld fs0, 0x00(sp) fld fs1, 0x08(sp) fld fs2, 0x10(sp) @@ -721,6 +757,7 @@ LOCAL_LABEL(DonePopping): fld fs9, 0x48(sp) fld fs10, 0x50(sp) fld fs11, 0x58(sp) +#endif FREE_CALL_FUNCLET_FRAME 0x60 EPILOG_RETURN @@ -774,7 +811,14 @@ LOCAL_LABEL(DonePopping): addi t3, a5, OFFSETOF__Thread__m_ThreadStateFlags addiw a6, zero, -17 // Mask value (0xFFFFFFEF) +#ifdef __riscv_atomic amoand.w a4, t3, a6 +#else + // No A extension: non-atomic read-modify-write. + lw a4, (t3) + and a4, a4, a6 + sw a4, (t3) +#endif // set preserved regs to the values expected by the funclet RESTORE_PRESERVED_REGISTERS a2 diff --git a/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S b/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S index 8522bc01037412..050770e34b0fc7 100644 --- a/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S +++ b/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S @@ -39,8 +39,10 @@ sd a2, 0x90(sp) # Save the FP return registers +#if __riscv_flen >= 64 fsd fa0, 0x98(sp) fsd fa1, 0xa0(sp) +#endif # Slot at sp+0xa8 is alignment padding # Perform the rest of the PInvokeTransitionFrame initialization. @@ -64,8 +66,10 @@ ld a2, 0x90(sp) // Restore the FP return registers +#if __riscv_flen >= 64 fld fa0, 0x98(sp) fld fa1, 0xa0(sp) +#endif // Restore callee saved registers EPILOG_RESTORE_REG_PAIR s1, s2, 0x20 diff --git a/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S b/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S index b38480f8113d2b..7e5fce8ced32ba 100644 --- a/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S +++ b/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S @@ -89,6 +89,7 @@ PROLOG_SAVE_REG_PAIR_INDEXED fp, ra, STACK_SIZE # Floating point registers +#if __riscv_flen >= 64 fsd fa0, FLOAT_ARG_OFFSET(sp) fsd fa1, FLOAT_ARG_OFFSET + 0x08(sp) fsd fa2, FLOAT_ARG_OFFSET + 0x10(sp) @@ -97,6 +98,7 @@ fsd fa5, FLOAT_ARG_OFFSET + 0x28(sp) fsd fa6, FLOAT_ARG_OFFSET + 0x30(sp) fsd fa7, FLOAT_ARG_OFFSET + 0x38(sp) +#endif # Space for return block data (0x10 bytes) @@ -113,6 +115,7 @@ #ifdef TRASH_SAVED_ARGUMENT_REGISTERS PREPARE_EXTERNAL_VAR RhpFpTrashValues, a1 +#if __riscv_flen >= 64 fld fa0, 0x00(a1) fld fa1, 0x08(a1) fld fa2, 0x10(a1) @@ -121,6 +124,7 @@ fld fa5, 0x28(a1) fld fa6, 0x30(a1) fld fa7, 0x38(a1) +#endif PREPARE_EXTERNAL_VAR RhpIntegerTrashValues, a1 @@ -143,6 +147,7 @@ ALTERNATE_ENTRY ReturnFrom\FunctionName mv t2, a0 # Restore floating point registers +#if __riscv_flen >= 64 fld fa0, FLOAT_ARG_OFFSET(sp) fld fa1, FLOAT_ARG_OFFSET + 0x08(sp) fld fa2, FLOAT_ARG_OFFSET + 0x10(sp) @@ -151,6 +156,7 @@ ALTERNATE_ENTRY ReturnFrom\FunctionName fld fa5, FLOAT_ARG_OFFSET + 0x28(sp) fld fa6, FLOAT_ARG_OFFSET + 0x30(sp) fld fa7, FLOAT_ARG_OFFSET + 0x38(sp) +#endif # Restore the argument registers ld a0, ARGUMENT_REGISTERS_OFFSET(sp) diff --git a/src/coreclr/pal/inc/unixasmmacrosriscv64.inc b/src/coreclr/pal/inc/unixasmmacrosriscv64.inc index 406074d6f49436..5342ef99465bf9 100644 --- a/src/coreclr/pal/inc/unixasmmacrosriscv64.inc +++ b/src/coreclr/pal/inc/unixasmmacrosriscv64.inc @@ -161,6 +161,9 @@ C_FUNC(\Name): // Reserve 64 bytes of memory before calling SAVE_FLOAT_ARGUMENT_REGISTERS .macro SAVE_FLOAT_ARGUMENT_REGISTERS reg, ofs + // No-op without an FPU; the caller reserves the slots either way, so frame + // offsets are unaffected. Same for the other FP sequences in this file. +#if __riscv_flen >= 64 fsd fa0, (\ofs)(\reg) fsd fa1, (\ofs + 8)(\reg) fsd fa2, (\ofs + 16)(\reg) @@ -169,6 +172,7 @@ C_FUNC(\Name): fsd fa5, (\ofs + 40)(\reg) fsd fa6, (\ofs + 48)(\reg) fsd fa7, (\ofs + 56)(\reg) +#endif .endm // Reserve 64 bytes of memory before calling SAVE_FLOAT_CALLEESAVED_REGISTERS @@ -199,6 +203,7 @@ C_FUNC(\Name): .endm .macro RESTORE_FLOAT_ARGUMENT_REGISTERS reg, ofs +#if __riscv_flen >= 64 fld fa0, (\ofs)(\reg) fld fa1, (\ofs + 8)(\reg) fld fa2, (\ofs + 16)(\reg) @@ -207,6 +212,7 @@ C_FUNC(\Name): fld fa5, (\ofs + 40)(\reg) fld fa6, (\ofs + 48)(\reg) fld fa7, (\ofs + 56)(\reg) +#endif .endm .macro RESTORE_FLOAT_CALLEESAVED_REGISTERS reg, ofs @@ -310,6 +316,7 @@ C_FUNC(\Name): // Save callee-saved floating point registers if requested (fs0-fs11) .if (__PWTB_PushCalleeSavedFloatRegs == 1) +#if __riscv_flen >= 64 fsd fs0, (__PWTB_FloatCalleeSavedRegisters)(sp) fsd fs1, (__PWTB_FloatCalleeSavedRegisters + 8)(sp) fsd fs2, (__PWTB_FloatCalleeSavedRegisters + 16)(sp) @@ -322,6 +329,7 @@ C_FUNC(\Name): fsd fs9, (__PWTB_FloatCalleeSavedRegisters + 72)(sp) fsd fs10, (__PWTB_FloatCalleeSavedRegisters + 80)(sp) fsd fs11, (__PWTB_FloatCalleeSavedRegisters + 88)(sp) +#endif .endif .endm @@ -405,6 +413,7 @@ C_FUNC(\Name): // Save FP callee-saved registers (fs0-fs11 = f8,f9,f18-f27) at offset 0 // RISC-V FP callee-saved: fs0=f8, fs1=f9, fs2-fs11=f18-f27 +#if __riscv_flen >= 64 fsd fs0, 0(sp) // f8 fsd fs1, 8(sp) // f9 fsd fs2, 16(sp) // f18 @@ -417,6 +426,7 @@ C_FUNC(\Name): fsd fs9, 72(sp) // f25 fsd fs10, 80(sp) // f26 fsd fs11, 88(sp) // f27 +#endif // Set target to TransitionBlock pointer addi \target, sp, 160 diff --git a/src/coreclr/pal/src/arch/riscv64/context2.S b/src/coreclr/pal/src/arch/riscv64/context2.S index 5bb06b0ecea702..7914eae9921244 100644 --- a/src/coreclr/pal/src/arch/riscv64/context2.S +++ b/src/coreclr/pal/src/arch/riscv64/context2.S @@ -26,6 +26,7 @@ LEAF_ENTRY RtlRestoreContext, _TEXT //64-bits FPR. addi t0, t4, CONTEXT_FPU_OFFSET +#if __riscv_flen >= 64 fld f0, (CONTEXT_F0)(t0) fld f1, (CONTEXT_F1)(t0) fld f2, (CONTEXT_F2)(t0) @@ -58,9 +59,12 @@ LEAF_ENTRY RtlRestoreContext, _TEXT fld f29, (CONTEXT_F29)(t0) fld f30, (CONTEXT_F30)(t0) fld f31, (CONTEXT_F31)(t0) +#endif lw t1, (CONTEXT_FLOAT_CONTROL_OFFSET)(t0) +#if __riscv_flen != 0 fscsr x0, t1 +#endif LOCAL_LABEL(No_Restore_CONTEXT_FLOATING_POINT): @@ -205,6 +209,7 @@ LOCAL_LABEL(Done_CONTEXT_INTEGER): addi a0, a0, CONTEXT_FPU_OFFSET +#if __riscv_flen >= 64 fsd f0, (CONTEXT_F0)(a0) fsd f1, (CONTEXT_F1)(a0) fsd f2, (CONTEXT_F2)(a0) @@ -237,8 +242,11 @@ LOCAL_LABEL(Done_CONTEXT_INTEGER): fsd f29, (CONTEXT_F29)(a0) fsd f30, (CONTEXT_F30)(a0) fsd f31, (CONTEXT_F31)(a0) +#endif +#if __riscv_flen != 0 frcsr t0 +#endif sd t0, (CONTEXT_FLOAT_CONTROL_OFFSET)(a0) LOCAL_LABEL(Done_CONTEXT_FLOATING_POINT): diff --git a/src/coreclr/runtime/riscv64/WriteBarriers.S b/src/coreclr/runtime/riscv64/WriteBarriers.S index c4a336be1caeb4..4dc36d2822d772 100644 --- a/src/coreclr/runtime/riscv64/WriteBarriers.S +++ b/src/coreclr/runtime/riscv64/WriteBarriers.S @@ -277,6 +277,7 @@ LEAF_END RhpAssignRef, _TEXT LEAF_ENTRY RhpCheckedLockCmpXchg LOCAL_LABEL(CmpXchgRetry): +#ifdef __riscv_atomic // Load the current value at the destination address. lr.d.aqrl t0, (a0) // t0 = *dest (load with acquire-release ordering) // Compare the loaded value with the comparand. @@ -285,6 +286,12 @@ LOCAL_LABEL(CmpXchgRetry): // Attempt to store the exchange value at the destination address. sc.d.rl t1, a1, (a0) // t1 = (store conditional result: 0 if successful, with release ordering) bnez t1, LOCAL_LABEL(CmpXchgRetry) // if store conditional failed, retry +#else + // No A extension: non-atomic compare-and-swap. + ld t0, (a0) + bne t0, a2, LOCAL_LABEL(CmpXchgNoUpdate) + sd a1, (a0) +#endif // See comment at the top of PalInterlockedOperationBarrier method for explanation why this memory // barrier is necessary. @@ -321,7 +328,13 @@ LEAF_END RhpCheckedLockCmpXchg // t1, t6: trashed // LEAF_ENTRY RhpCheckedXchg +#ifdef __riscv_atomic amoswap.d.aqrl t1, a1, (a0) +#else + // No A extension: non-atomic exchange, old value in t1. + ld t1, (a0) + sd a1, (a0) +#endif // See comment at the top of PalInterlockedOperationBarrier method for explanation why this memory // barrier is necessary. diff --git a/src/coreclr/tools/Common/CommandLineHelpers.cs b/src/coreclr/tools/Common/CommandLineHelpers.cs index 58ff170a7e5252..0d917dc1eecac8 100644 --- a/src/coreclr/tools/Common/CommandLineHelpers.cs +++ b/src/coreclr/tools/Common/CommandLineHelpers.cs @@ -27,6 +27,8 @@ internal static partial class Helpers public static string[] ValidOS { get; } = ["windows", "linux", "freebsd", "openbsd", "osx", "maccatalyst", "ios", "iossimulator", "tvos", "tvossimulator", "android", "browser", "wasi"]; public static string[] ValidArchitectures { get; } = ["arm", "armel", "arm64", "x86", "x64", "riscv64", "loongarch64", "wasm"]; + // Targets that only the NativeAOT compiler supports (no ReadyToRun): the RISC-V lp64 soft-float ABI. + public static string[] ValidArchitecturesNativeAot { get; } = [.. ValidArchitectures, "riscv64-lp64"]; public static Dictionary BuildPathDictionary(IReadOnlyList tokens, bool strict) { @@ -119,7 +121,7 @@ public static TargetArchitecture GetTargetArchitecture(string token) "arm64" => TargetArchitecture.ARM64, "wasm" => TargetArchitecture.Wasm32, "loongarch64" => TargetArchitecture.LoongArch64, - "riscv64" => TargetArchitecture.RiscV64, + "riscv64" or "riscv64-lp64" => TargetArchitecture.RiscV64, _ => throw new CommandLineException($"Target architecture '{token}' is not supported") }; } @@ -136,6 +138,7 @@ public static (TargetArchitecture, TargetOS, TargetAbi) GetTargetSpec(string tar { (_, "armel") => TargetAbi.NativeAotArmel, ("android", "arm") => TargetAbi.NativeAotArmel, + (_, "riscv64-lp64") => TargetAbi.NativeAotRiscV64SoftFloat, _ => TargetAbi.NativeAot, }; diff --git a/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs b/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs index 3e252c360be5ea..ad48f024bc5e54 100644 --- a/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs +++ b/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs @@ -41,6 +41,7 @@ internal sealed partial class ElfObjectWriter : UnixObjectWriter private readonly bool _useInlineRelocationAddends; private readonly ushort _machine; private readonly bool _useSoftFPAbi; + private readonly uint _riscV64ElfFlags; private readonly List _sections = new(); private readonly List _symbols = new(); private uint _localSymbolCount; @@ -53,6 +54,25 @@ internal sealed partial class ElfObjectWriter : UnixObjectWriter private static readonly ObjectNodeSection ArmTextThunkSection = new ObjectNodeSection(".text.thunks", SectionType.Executable); private static readonly ObjectNodeSection CommentSection = new ObjectNodeSection(".comment", SectionType.ReadOnly); + /// + /// RISC-V ELF header flags: the floating-point ABI comes from the target ABI + /// (EF_RISCV_FLOAT_ABI_DOUBLE for lp64d, EF_RISCV_FLOAT_ABI_SOFT for lp64 - the + /// linker rejects objects with mixed floating-point ABIs), EF_RISCV_RVC is set + /// when the target has the C extension. + /// + internal static uint GetRiscV64ElfFlags(TargetAbi abi, ObjectWritingOptions options) + { + const uint EF_RISCV_RVC = 0x0001; + const uint EF_RISCV_FLOAT_ABI_DOUBLE = 0x0004; + + uint flags = abi == TargetAbi.NativeAotRiscV64SoftFloat ? 0u : EF_RISCV_FLOAT_ABI_DOUBLE; + if (options.HasFlag(ObjectWritingOptions.RiscV64Compressed)) + { + flags |= EF_RISCV_RVC; + } + return flags; + } + public ElfObjectWriter(NodeFactory factory, ObjectWritingOptions options) : base(factory, options) { @@ -68,6 +88,7 @@ public ElfObjectWriter(NodeFactory factory, ObjectWritingOptions options) }; _useInlineRelocationAddends = _machine is EM_386 or EM_ARM; _useSoftFPAbi = _machine is EM_ARM && factory.Target.Abi == TargetAbi.NativeAotArmel; + _riscV64ElfFlags = GetRiscV64ElfFlags(factory.Target.Abi, options); // By convention the symbol table starts with empty symbol _symbols.Add(new ElfSymbol {}); @@ -765,7 +786,7 @@ private void EmitObjectFile(Stream outputFileStream) { EM_ARM => 0x05000000u, // For ARM32 claim conformance with the EABI specification EM_LOONGARCH => 0x43u, // For LoongArch ELF psABI specify the ABI version (1) and modifiers (64-bit GPRs, 64-bit FPRs) - EM_RISCV => 0x0005u, // EF_RISCV_RVC (RVC ABI) | EF_RISCV_FLOAT_ABI_DOUBLE (double precision floating-point ABI). + EM_RISCV => _riscV64ElfFlags, _ => 0u }, }; diff --git a/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs b/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs index 61276810208aca..7e9f4e61f197a5 100644 --- a/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs +++ b/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs @@ -13,5 +13,9 @@ public enum ObjectWritingOptions ControlFlowGuard = 0x02, UseDwarf5 = 0x4, GenerateUnwindInfo = 0x8, + /// + /// RISC-V: the target supports the C (compressed instructions) extension + /// + RiscV64Compressed = 0x10, } } diff --git a/src/coreclr/tools/Common/InstructionSetHelpers.cs b/src/coreclr/tools/Common/InstructionSetHelpers.cs index ed7f1649449091..5240af5130fdae 100644 --- a/src/coreclr/tools/Common/InstructionSetHelpers.cs +++ b/src/coreclr/tools/Common/InstructionSetHelpers.cs @@ -18,7 +18,7 @@ namespace System.CommandLine internal static partial class Helpers { public static InstructionSetSupport ConfigureInstructionSetSupport(string instructionSet, int maxVectorTBitWidth, bool isVectorTOptimistic, TargetArchitecture targetArchitecture, TargetOS targetOS, - string mustNotBeMessage, string invalidImplicationMessage, Logger logger, bool allowOptimistic, bool isReadyToRun) + string mustNotBeMessage, string invalidImplicationMessage, Logger logger, bool allowOptimistic, bool isReadyToRun, TargetAbi targetAbi) { InstructionSetSupportBuilder instructionSetSupportBuilder = new(targetArchitecture); @@ -94,6 +94,21 @@ public static InstructionSetSupport ConfigureInstructionSetSupport(string instru instructionSetSupportBuilder.AddSupportedInstructionSet("base"); instructionSetSupportBuilder.AddSupportedInstructionSet("simd128"); } + else if (targetArchitecture == TargetArchitecture.RiscV64) + { + // The rv64gc baseline: D implies F, so "d", "c" and "a" cover the G+C + // extensions. The lp64 (soft-float) ABI target has no F/D by definition; + // it still defaults to C and A, and a reduced-ISA target drops those with + // --instruction-set=-a,-c. Dropping A also requires ilc's + // --assume-no-concurrency, because Interlocked is not atomic without it. + instructionSetSupportBuilder.AddSupportedInstructionSet("base"); + if (targetAbi != TargetAbi.NativeAotRiscV64SoftFloat) + { + instructionSetSupportBuilder.AddSupportedInstructionSet("d"); + } + instructionSetSupportBuilder.AddSupportedInstructionSet("c"); + instructionSetSupportBuilder.AddSupportedInstructionSet("a"); + } bool throttleAvx512 = false; diff --git a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs index 5ff98b8f980021..5f21a5c8866761 100644 --- a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs +++ b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs @@ -409,6 +409,22 @@ public enum ReadyToRunHelper TypeHandleToRuntimeType, GetRefAny, TypeHandleToRuntimeTypeHandle, + + // Soft-float arithmetic (targets without an FPU). NativeAOT only. + FltAdd, + FltSub, + FltMul, + FltDiv, + DblAdd, + DblSub, + DblMul, + DblDiv, + FltCmpLe, + FltCmpGe, + DblCmpLe, + DblCmpGe, + Flt2Dbl, + Dbl2Flt, } // Enum used for HFA type recognition. diff --git a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs index 4c5574d6870c32..efb7302babd894 100644 --- a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs +++ b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs @@ -106,5 +106,9 @@ public enum ReadyToRunInstructionSet Cssc = 93, Zicond = 94, Fp16 = 95, + RiscV64F = 96, + RiscV64D = 97, + RiscV64C = 98, + RiscV64A = 99, } } diff --git a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs index 632e3053189a72..28d370e346280c 100644 --- a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs +++ b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs @@ -77,6 +77,10 @@ public static class ReadyToRunInstructionSetHelper case InstructionSet.RiscV64_Zbb: return ReadyToRunInstructionSet.Zbb; case InstructionSet.RiscV64_Zbs: return ReadyToRunInstructionSet.Zbs; case InstructionSet.RiscV64_Zicond: return ReadyToRunInstructionSet.Zicond; + case InstructionSet.RiscV64_F: return ReadyToRunInstructionSet.RiscV64F; + case InstructionSet.RiscV64_D: return ReadyToRunInstructionSet.RiscV64D; + case InstructionSet.RiscV64_C: return ReadyToRunInstructionSet.RiscV64C; + case InstructionSet.RiscV64_A: return ReadyToRunInstructionSet.RiscV64A; default: throw new Exception("Unknown instruction set"); } diff --git a/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs b/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs index d874da3451ae71..6eddb38bcdb66d 100644 --- a/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs +++ b/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs @@ -41,6 +41,22 @@ public enum CorInfoHelpFunc CORINFO_HELP_FLTREM, CORINFO_HELP_DBLREM, + // Soft-float helpers (targets without an FPU); see corinfo.h + CORINFO_HELP_FLTADD, + CORINFO_HELP_FLTSUB, + CORINFO_HELP_FLTMUL, + CORINFO_HELP_FLTDIV, + CORINFO_HELP_DBLADD, + CORINFO_HELP_DBLSUB, + CORINFO_HELP_DBLMUL, + CORINFO_HELP_DBLDIV, + CORINFO_HELP_FLTCMP_LE, + CORINFO_HELP_FLTCMP_GE, + CORINFO_HELP_DBLCMP_LE, + CORINFO_HELP_DBLCMP_GE, + CORINFO_HELP_FLT2DBL, + CORINFO_HELP_DBL2FLT, + /* Allocating a new object. Always use ICorClassInfo::getNewHelper() to decide which is the right helper to use to allocate an object of a given type. */ diff --git a/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs b/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs index 615f5c91ccf897..0048f8a4abadd2 100644 --- a/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs +++ b/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs @@ -4915,6 +4915,13 @@ private uint getJitFlags(ref CORJIT_FLAGS flags, uint sizeInBytes) flags.Set(CorJitFlag.CORJIT_FLAG_SOFTFP_ABI); } + if (this.MethodBeingCompiled.Context.Target.Abi == TargetAbi.NativeAotRiscV64SoftFloat) + { + // RISC-V lp64: FP values are passed in integer registers and the FP + // arithmetic goes through the soft-float helpers. + flags.Set(CorJitFlag.CORJIT_FLAG_SOFTFP_ABI); + } + if (this.MethodBeingCompiled.IsAsyncCall() #if !READYTORUN || (_compilation.TypeSystemContext.IsSpecialUnboxingThunk(this.MethodBeingCompiled) && _compilation.TypeSystemContext.GetTargetOfSpecialUnboxingThunk(this.MethodBeingCompiled).IsAsyncCall()) diff --git a/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs b/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs index 870544e2156425..4feeb6db799930 100644 --- a/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs +++ b/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs @@ -63,6 +63,10 @@ public enum InstructionSet RiscV64_Zbb = InstructionSet_RiscV64.Zbb, RiscV64_Zbs = InstructionSet_RiscV64.Zbs, RiscV64_Zicond = InstructionSet_RiscV64.Zicond, + RiscV64_F = InstructionSet_RiscV64.F, + RiscV64_D = InstructionSet_RiscV64.D, + RiscV64_C = InstructionSet_RiscV64.C, + RiscV64_A = InstructionSet_RiscV64.A, Wasm32_WasmBase = InstructionSet_Wasm32.WasmBase, Wasm32_PackedSimd = InstructionSet_Wasm32.PackedSimd, Wasm32_Vector128 = InstructionSet_Wasm32.Vector128, @@ -215,6 +219,10 @@ public enum InstructionSet_RiscV64 Zbb = 3, Zbs = 4, Zicond = 5, + F = 6, + D = 7, + C = 8, + A = 9, } public enum InstructionSet_Wasm32 @@ -615,6 +623,14 @@ public static InstructionSetFlags ExpandInstructionSetByImplicationHelper(Target resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); if (resultflags.HasInstructionSet(InstructionSet.RiscV64_Zicond)) resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_F)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_D)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_F); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_C)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_A)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); break; case TargetArchitecture.Wasm32: @@ -926,6 +942,14 @@ private static InstructionSetFlags ExpandInstructionSetByReverseImplicationHelpe resultflags.AddInstructionSet(InstructionSet.RiscV64_Zbs); if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) resultflags.AddInstructionSet(InstructionSet.RiscV64_Zicond); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_F); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_F)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_D); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_C); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_A); break; case TargetArchitecture.Wasm32: @@ -1188,6 +1212,10 @@ public static IEnumerable ArchitectureToValidInstructionSets yield return new InstructionSetInfo("zbb", "", InstructionSet.RiscV64_Zbb, true); yield return new InstructionSetInfo("zbs", "", InstructionSet.RiscV64_Zbs, true); yield return new InstructionSetInfo("zicond", "", InstructionSet.RiscV64_Zicond, true); + yield return new InstructionSetInfo("f", "", InstructionSet.RiscV64_F, true); + yield return new InstructionSetInfo("d", "", InstructionSet.RiscV64_D, true); + yield return new InstructionSetInfo("c", "", InstructionSet.RiscV64_C, true); + yield return new InstructionSetInfo("a", "", InstructionSet.RiscV64_A, true); break; case TargetArchitecture.Wasm32: diff --git a/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt b/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt index a70cd7d9991687..dd252841739d62 100644 --- a/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt +++ b/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt @@ -26,7 +26,7 @@ ; DO NOT CHANGE R2R NUMERIC VALUES OF THE EXISTING SETS. Changing R2R numeric values definitions would be R2R format breaking change. ; The ISA definitions should also be mapped to `hwintrinsicIsaRangeArray` in hwintrinsic.cpp. -; NEXT_AVAILABLE_R2R_BIT = 96 +; NEXT_AVAILABLE_R2R_BIT = 100 ; Definition of X86 instruction sets definearch ,X86 ,32Bit ,X64, X64, X86 @@ -287,11 +287,19 @@ instructionset ,RiscV64 , ,Zba ,57 ,Zba ,zba instructionset ,RiscV64 , ,Zbb ,58 ,Zbb ,zbb instructionset ,RiscV64 , ,Zbs ,84 ,Zbs ,zbs instructionset ,RiscV64 , ,Zicond ,94 ,Zicond ,zicond +instructionset ,RiscV64 , ,RiscV64F ,96 ,F ,f +instructionset ,RiscV64 , ,RiscV64D ,97 ,D ,d +instructionset ,RiscV64 , ,RiscV64C ,98 ,C ,c +instructionset ,RiscV64 , ,RiscV64A ,99 ,A ,a implication ,RiscV64 ,Zbb ,RiscV64Base implication ,RiscV64 ,Zba ,RiscV64Base implication ,RiscV64 ,Zbs ,RiscV64Base implication ,RiscV64 ,Zicond ,RiscV64Base +implication ,RiscV64 ,F ,RiscV64Base +implication ,RiscV64 ,D ,F +implication ,RiscV64 ,C ,RiscV64Base +implication ,RiscV64 ,A ,RiscV64Base ; ,name and aliases ,archs ,lower baselines included by implication instructionsetgroup ,x86-64-v2 ,X64 X86 ,base diff --git a/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs b/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs index 4c9862cd0c8698..2ceac0f2363a91 100644 --- a/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs +++ b/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs @@ -39,6 +39,11 @@ public enum TargetAbi /// model for armel execution model /// NativeAotArmel, + /// + /// RISC-V lp64 (soft-float) execution model: no F/D extensions, floating-point + /// values are passed in integer registers and computed by the soft-float helpers + /// + NativeAotRiscV64SoftFloat, } /// diff --git a/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs b/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs index 99fc77b6145ac3..455f30df1e021c 100644 --- a/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs +++ b/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs @@ -219,6 +219,50 @@ public static void GetEntryPoint(TypeSystemContext context, ReadyToRunHelper id, mangledName = "fmodf"; break; + // Soft-float arithmetic: the compiler-rt/libgcc builtins of the target toolchain + case ReadyToRunHelper.FltAdd: + mangledName = "__addsf3"; + break; + case ReadyToRunHelper.FltSub: + mangledName = "__subsf3"; + break; + case ReadyToRunHelper.FltMul: + mangledName = "__mulsf3"; + break; + case ReadyToRunHelper.FltDiv: + mangledName = "__divsf3"; + break; + case ReadyToRunHelper.DblAdd: + mangledName = "__adddf3"; + break; + case ReadyToRunHelper.DblSub: + mangledName = "__subdf3"; + break; + case ReadyToRunHelper.DblMul: + mangledName = "__muldf3"; + break; + case ReadyToRunHelper.DblDiv: + mangledName = "__divdf3"; + break; + case ReadyToRunHelper.FltCmpLe: + mangledName = "__lesf2"; + break; + case ReadyToRunHelper.FltCmpGe: + mangledName = "__gesf2"; + break; + case ReadyToRunHelper.DblCmpLe: + mangledName = "__ledf2"; + break; + case ReadyToRunHelper.DblCmpGe: + mangledName = "__gedf2"; + break; + case ReadyToRunHelper.Flt2Dbl: + mangledName = "__extendsfdf2"; + break; + case ReadyToRunHelper.Dbl2Flt: + mangledName = "__truncdfsf2"; + break; + case ReadyToRunHelper.LMul: mangledName = "RhpLMul"; break; diff --git a/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs b/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs index 86fdf190fd18cd..eb0d7b924b1030 100644 --- a/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs +++ b/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs @@ -128,6 +128,7 @@ private static PerfmapTokensForTarget TranslateTargetDetailsToPerfmapConstants(T TargetAbi.Unknown => PerfMapAbiToken.Unknown, TargetAbi.NativeAot => PerfMapAbiToken.Default, TargetAbi.NativeAotArmel => PerfMapAbiToken.Armel, + TargetAbi.NativeAotRiscV64SoftFloat => PerfMapAbiToken.Default, _ => throw new NotImplementedException(details.Abi.ToString()) }; diff --git a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs index 827907a5aa43f3..1182849b6b3fd5 100644 --- a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs +++ b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs @@ -66,6 +66,29 @@ static void Validate(ReadyToRunReader reader) } } + [Fact] + public void RiscV64SoftFloatTargetIsRejected() + { + // The riscv64-lp64 (soft-float) ABI is NativeAOT-only: crossgen2 rejects it on the + // command line instead of compiling with helpers that have no ReadyToRun encoding. + var module = new CompiledAssembly + { + AssemblyName = nameof(RiscV64SoftFloatTargetIsRejected), + SourceResourceNames = ["ThumbBit/HotColdSplitting.cs"], + }; + + new R2RTestRunner(_output).Run(new R2RTestCase( + nameof(RiscV64SoftFloatTargetIsRejected), + [ + new(nameof(RiscV64SoftFloatTargetIsRejected), [new CrossgenAssembly(module)]) + { + TargetOS = "linux", + TargetArchitecture = "riscv64-lp64", + ExpectedFailure = "is not supported by ReadyToRun", + }, + ])); + } + [Fact] public void GenericTypeConstraintsAllowVariantParameters() { diff --git a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs index 4fdb38bc20efbf..11b18769ae942b 100644 --- a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs +++ b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs @@ -94,6 +94,21 @@ internal sealed class CrossgenCompilation(string name, List as /// public Action? Validate { get; init; } + /// + /// When set, crossgen2 is expected to fail and its standard error must contain this text. + /// + public string? ExpectedFailure { get; init; } + + /// + /// Overrides the target OS passed to crossgen2; defaults to the test run's target. + /// + public string? TargetOS { get; init; } + + /// + /// Overrides the target architecture passed to crossgen2; defaults to the test run's target. + /// + public string? TargetArchitecture { get; init; } + public string Name => name; public bool IsComposite => Options.Contains(Crossgen2Option.Composite); @@ -309,7 +324,7 @@ private static string RunCrossgenCompilation( foreach (var option in compilation.Options) args.Add(option.ToArg()); - args.AddRange(["--targetos", TestPaths.TargetOS, "--targetarch", TestPaths.TargetArchitecture]); + args.AddRange(["--targetos", compilation.TargetOS ?? TestPaths.TargetOS, "--targetarch", compilation.TargetArchitecture ?? TestPaths.TargetArchitecture]); // Caller-supplied raw args (for options that take values, e.g. --determinism-stress=N) args.AddRange(compilation.AdditionalArgs); @@ -324,6 +339,12 @@ private static string RunCrossgenCompilation( args.Add($"--out"); args.Add($"{outputFile}"); var result = driver.Compile(args); + if (compilation.ExpectedFailure is string expectedFailure) + { + Assert.False(result.Success, $"crossgen2 unexpectedly succeeded for '{testName}'"); + Assert.Contains(expectedFailure, result.StandardError); + return outputFile; + } Assert.True(result.Success, $"crossgen2 failed for '{testName}':\n{result.StandardError}\n{result.StandardOutput}"); diff --git a/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs b/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs index e812ae60c3e2f3..9c3e87178be27b 100644 --- a/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs +++ b/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs @@ -60,6 +60,26 @@ internal RyuJitCompilation( _compilationOptions = options; InstructionSetSupport = instructionSetSupport; + if (nodeFactory.Target.Architecture == TargetArchitecture.RiscV64) + { + // The ABI is a property of the target, not of the instruction set: opting out of + // F/D does not make the runtime and the native libraries lp64, and the lp64 + // target has no F/D by definition. The lp64f ABI (F without D) is not supported. + bool hasF = instructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_F); + bool hasD = instructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_D); + if (nodeFactory.Target.Abi == TargetAbi.NativeAotRiscV64SoftFloat) + { + if (hasF || hasD) + throw new NotSupportedException( + "The riscv64-lp64 (soft-float) ABI target has no F and D extensions; remove them from the instruction set."); + } + else if (!hasF || !hasD) + { + throw new NotSupportedException( + "The riscv64 lp64d ABI requires the F and D extensions; targets without them must use the riscv64-lp64 (soft-float) ABI."); + } + } + _profileDataManager = profileDataManager; _methodImportationErrorProvider = errorProvider; @@ -126,6 +146,10 @@ protected override void CompileInternal(string outputFile, ObjectDumper dumper) if ((_compilationOptions & RyuJitCompilationOptions.ControlFlowGuardAnnotations) != 0) options |= ObjectWritingOptions.ControlFlowGuard; + if ((NodeFactory.Target.Architecture == TargetArchitecture.RiscV64) + && InstructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_C)) + options |= ObjectWritingOptions.RiscV64Compressed; + ObjectWriter.ObjectWriter.EmitObject(outputFile, nodes, NodeFactory, options, dumper, _logger); } diff --git a/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs b/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs index c17dae1393b11f..f5cc12cf61694c 100644 --- a/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs +++ b/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs @@ -737,6 +737,48 @@ private ISymbolNode GetHelperFtnUncached(CorInfoHelpFunc ftnNum) case CorInfoHelpFunc.CORINFO_HELP_DBLREM: id = ReadyToRunHelper.DblRem; break; + case CorInfoHelpFunc.CORINFO_HELP_FLTADD: + id = ReadyToRunHelper.FltAdd; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTSUB: + id = ReadyToRunHelper.FltSub; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTMUL: + id = ReadyToRunHelper.FltMul; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTDIV: + id = ReadyToRunHelper.FltDiv; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLADD: + id = ReadyToRunHelper.DblAdd; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLSUB: + id = ReadyToRunHelper.DblSub; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLMUL: + id = ReadyToRunHelper.DblMul; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLDIV: + id = ReadyToRunHelper.DblDiv; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTCMP_LE: + id = ReadyToRunHelper.FltCmpLe; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTCMP_GE: + id = ReadyToRunHelper.FltCmpGe; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLCMP_LE: + id = ReadyToRunHelper.DblCmpLe; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLCMP_GE: + id = ReadyToRunHelper.DblCmpGe; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLT2DBL: + id = ReadyToRunHelper.Flt2Dbl; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBL2FLT: + id = ReadyToRunHelper.Dbl2Flt; + break; case CorInfoHelpFunc.CORINFO_HELP_JIT_PINVOKE_BEGIN: id = ReadyToRunHelper.PInvokeBegin; diff --git a/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs b/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs index 875bbeaf73b9e7..f2e47cf6012b1a 100644 --- a/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs +++ b/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs @@ -118,6 +118,8 @@ internal sealed class ILCompilerRootCommand : RootCommand new("--parallelism") { CustomParser = MakeParallelism, DefaultValueFactory = MakeParallelism, Description = "Maximum number of threads to use during compilation" }; public Option InstructionSet { get; } = new("--instruction-set") { Description = "Instruction set to allow or disallow" }; + public Option AssumeNoConcurrency { get; } = + new("--assume-no-concurrency") { Description = "RISC-V: assert that the target runs on one hart and is never preempted. Required to build without the A extension, because Interlocked then lowers to a non-atomic read/modify/write" }; public Option MaxVectorTBitWidth { get; } = new("--max-vectort-bitwidth") { Description = "Maximum width, in bits, that Vector is allowed to be" }; public Option Guard { get; } = @@ -243,6 +245,7 @@ public ILCompilerRootCommand(string[] args) : base(".NET Native IL Compiler") Options.Add(RuntimeKnobs); Options.Add(Parallelism); Options.Add(InstructionSet); + Options.Add(AssumeNoConcurrency); Options.Add(MaxVectorTBitWidth); Options.Add(Guard); Options.Add(Dehydrate); @@ -352,7 +355,7 @@ public static void PrintExtendedHelp(ParseResult _) Console.WriteLine("Valid switches for {0} are: '{1}'. The default value is '{2}'\n", "--targetos", string.Join("', '", Helpers.ValidOS), Helpers.GetTargetOS(null).ToString().ToLowerInvariant()); - Console.WriteLine(string.Format("Valid switches for {0} are: '{1}'. The default value is '{2}'\n", "--targetarch", string.Join("', '", Helpers.ValidArchitectures), Helpers.GetTargetArchitecture(null).ToString().ToLowerInvariant())); + Console.WriteLine(string.Format("Valid switches for {0} are: '{1}'. The default value is '{2}'\n", "--targetarch", string.Join("', '", Helpers.ValidArchitecturesNativeAot), Helpers.GetTargetArchitecture(null).ToString().ToLowerInvariant())); Console.WriteLine("The allowable values for the --instruction-set option are described in the table below. Each architecture has a different set of valid " + "instruction sets, and multiple instruction sets may be specified by separating the instructions sets by a ','. By default other instruction sets not " + @@ -360,7 +363,7 @@ public static void PrintExtendedHelp(ParseResult _) "All such light-up can be disallowed by specifying '-optimistic'. The instruction sets supported by the machine invoking the tool can be targeted by " + "specifying 'native'. For example 'native', 'avx,aes', 'avx,aes,-avx2', or 'avx,aes,-optimistic'"); - foreach (string arch in Helpers.ValidArchitectures) + foreach (string arch in Helpers.ValidArchitecturesNativeAot) { TargetArchitecture targetArch = Helpers.GetTargetArchitecture(arch); bool first = true; diff --git a/src/coreclr/tools/aot/ILCompiler/Program.cs b/src/coreclr/tools/aot/ILCompiler/Program.cs index 34305bc9ebf504..4dab289a37afda 100644 --- a/src/coreclr/tools/aot/ILCompiler/Program.cs +++ b/src/coreclr/tools/aot/ILCompiler/Program.cs @@ -109,7 +109,20 @@ public int Run() InstructionSetSupport instructionSetSupport = Helpers.ConfigureInstructionSetSupport(Get(_command.InstructionSet), Get(_command.MaxVectorTBitWidth), isVectorTOptimistic, targetArchitecture, targetOS, "Unrecognized instruction set {0}", "Unsupported combination of instruction sets: {0}/{1}", logger, allowOptimistic: _command.OptimizationMode != OptimizationMode.PreferSize, - isReadyToRun: false); + isReadyToRun: false, + targetAbi: targetAbi); + + if (targetArchitecture == TargetArchitecture.RiscV64 && + !instructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_A) && + !Get(_command.AssumeNoConcurrency)) + { + // Without A there is no atomic memory operation to emit, so Interlocked and + // CmpXchg lower to a plain read/modify/write. Whether that is sufficient is a + // property of the execution environment, not of the ISA or the ABI, so it has + // to be asserted rather than inferred. + throw new CommandLineException( + "Building without the A extension requires --assume-no-concurrency (one hart, no preemption): Interlocked operations are not atomic without it."); + } string systemModuleName = Get(_command.SystemModuleName); string reflectionData = Get(_command.ReflectionData); diff --git a/src/coreclr/tools/aot/crossgen2/Program.cs b/src/coreclr/tools/aot/crossgen2/Program.cs index ca0c40ac43b632..cb6203c5d0539f 100644 --- a/src/coreclr/tools/aot/crossgen2/Program.cs +++ b/src/coreclr/tools/aot/crossgen2/Program.cs @@ -80,6 +80,11 @@ public int Run() (TargetArchitecture targetArchitecture, TargetOS targetOS, TargetAbi targetAbi) = Helpers.GetTargetSpec(Get(_command.TargetArchitecture), Get(_command.TargetOS)); + if (targetAbi == TargetAbi.NativeAotRiscV64SoftFloat) + { + // The soft-float helpers have no ReadyToRun encoding; the target is NativeAOT only. + throw new CommandLineException($"Target architecture '{Get(_command.TargetArchitecture)}' is not supported by ReadyToRun"); + } // The portable call-helpers generator is currently supported only for Wasm. if (_generatePortableCallHelpers is not null @@ -108,7 +113,8 @@ public int Run() InstructionSetSupport instructionSetSupport = Helpers.ConfigureInstructionSetSupport(Get(_command.InstructionSet), Get(_command.MaxVectorTBitWidth), isVectorTOptimistic, targetArchitecture, targetOS, SR.InstructionSetMustNotBe, SR.InstructionSetInvalidImplication, logger, allowOptimistic: allowOptimistic, - isReadyToRun: true); + isReadyToRun: true, + targetAbi: targetAbi); if (!targetAllowsRuntimeCodeGeneration) { instructionSetSupport = Helpers.GetFixedInstructionSetSupport(instructionSetSupport); diff --git a/src/coreclr/vm/codeman.cpp b/src/coreclr/vm/codeman.cpp index 4e4aa6f787a0b5..335a38f863306b 100644 --- a/src/coreclr/vm/codeman.cpp +++ b/src/coreclr/vm/codeman.cpp @@ -1789,6 +1789,24 @@ void EEJitManager::SetCpuInfo() if (g_pConfig->EnableHWIntrinsic()) { CPUCompileFlags.Set(InstructionSet_RiscV64Base); + + // F/D/C/A belong to the rv64gc baseline and are not reported through + // hwprobe; the floor is the ISA this runtime itself was built for. +#if defined(__riscv_flen) && __riscv_flen >= 32 + CPUCompileFlags.Set(InstructionSet_F); +#endif +#if defined(__riscv_flen) && __riscv_flen >= 64 + CPUCompileFlags.Set(InstructionSet_D); +#endif +#ifdef __riscv_compressed + if (CLRConfig::GetConfigValue(CLRConfig::EXTERNAL_EnableRiscV64Compressed)) + { + CPUCompileFlags.Set(InstructionSet_C); + } +#endif +#ifdef __riscv_atomic + CPUCompileFlags.Set(InstructionSet_A); +#endif } if (((cpuFeatures & RiscV64IntrinsicConstants_Zba) != 0) && CLRConfig::GetConfigValue(CLRConfig::EXTERNAL_EnableRiscV64Zba)) diff --git a/src/coreclr/vm/precode.h b/src/coreclr/vm/precode.h index 2b6aef0db9251a..ff5b21fa0a12be 100644 --- a/src/coreclr/vm/precode.h +++ b/src/coreclr/vm/precode.h @@ -394,7 +394,13 @@ struct FixupPrecode static const int FixupCodeOffset = 12; #elif defined(TARGET_RISCV64) static const SIZE_T CodeSize = 32; + // The thunk's second half starts after the tail jump, which is 2 bytes + // shorter with the C extension. Keep in sync with thunktemplates.S. +#ifdef __riscv_compressed static const int FixupCodeOffset = 10; +#else + static const int FixupCodeOffset = 12; +#endif #endif // TARGET_AMD64 BYTE m_code[CodeSize]; diff --git a/src/coreclr/vm/riscv64/asmhelpers.S b/src/coreclr/vm/riscv64/asmhelpers.S index 486fce7790ec56..6e94858f41e277 100644 --- a/src/coreclr/vm/riscv64/asmhelpers.S +++ b/src/coreclr/vm/riscv64/asmhelpers.S @@ -489,8 +489,10 @@ NESTED_ENTRY OnHijackTripThread, _TEXT, NoHandler sd a2, 136(sp) // save any FP/HFA return value(s) +#if __riscv_flen >= 64 fsd f0, 144(sp) fsd f1, 152(sp) +#endif addi a0, sp, 0 call C_FUNC(OnHijackWorker) @@ -504,8 +506,10 @@ NESTED_ENTRY OnHijackTripThread, _TEXT, NoHandler ld a2, 136(sp) // restore any FP/HFA return value(s) +#if __riscv_flen >= 64 fld f0, 144(sp) fld f1, 152(sp) +#endif EPILOG_RESTORE_REG_PAIR s1, s2, 16 EPILOG_RESTORE_REG_PAIR s3, s4, 32 @@ -1206,7 +1210,9 @@ NESTED_ENTRY CallJittedMethodRetDouble, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen >= 64 fsd fa0, 0(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1230,7 +1236,9 @@ NESTED_ENTRY CallJittedMethodRetFloat, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen != 0 fsw fa0, 0(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1279,8 +1287,10 @@ NESTED_ENTRY CallJittedMethodRet2Double, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen >= 64 fsd fa0, 0(a2) fsd fa1, 8(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1304,8 +1314,10 @@ NESTED_ENTRY CallJittedMethodRet2Float, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen != 0 fsw fa0, 0(a2) fsw fa1, 4(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1329,7 +1341,9 @@ NESTED_ENTRY CallJittedMethodRetFloatInt, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen >= 64 fsd fa0, 0(a2) +#endif sd a0, 8(a2) EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 @@ -1355,7 +1369,9 @@ NESTED_ENTRY CallJittedMethodRetIntFloat, _TEXT, NoHandler sd a2, 0(a4) ld a2, 16(fp) sd a0, 0(a2) +#if __riscv_flen >= 64 fsd fa0, 8(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1434,7 +1450,9 @@ NESTED_ENTRY InterpreterStubRetDouble, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen >= 64 fld fa0, 0(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRetDouble, _TEXT @@ -1470,8 +1488,10 @@ NESTED_ENTRY InterpreterStubRet2Double, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen >= 64 fld fa0, 0(a0) fld fa1, 8(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRet2Double, _TEXT @@ -1483,7 +1503,9 @@ NESTED_ENTRY InterpreterStubRetFloat, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen != 0 flw fa0, 0(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRetFloat, _TEXT @@ -1495,8 +1517,10 @@ NESTED_ENTRY InterpreterStubRet2Float, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen != 0 flw fa0, 0(a0) flw fa1, 4(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRet2Float, _TEXT @@ -1508,7 +1532,9 @@ NESTED_ENTRY InterpreterStubRetFloatInt, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen >= 64 fld fa0, 0(a0) +#endif ld a0, 8(a0) EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN @@ -1522,7 +1548,9 @@ NESTED_ENTRY InterpreterStubRetIntFloat, _TEXT, NoHandler mv a2, zero call C_FUNC(ExecuteInterpretedMethod) ld a1, 0(a0) +#if __riscv_flen >= 64 fld fa0, 8(a0) +#endif mv a0, a1 EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN @@ -1979,9 +2007,14 @@ ALTERNATE_ENTRY Store_A7 LEAF_END Store_A1_A2_A3_A4_A5_A6_A7 // Float point load/store routines +// +// The symbols are always defined because the call stub generator references +// them; without an FPU nothing is routed through fa0-fa7, so they are unused. LEAF_ENTRY Load_FA0 +#if __riscv_flen >= 64 fld fa0, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -1989,8 +2022,10 @@ LEAF_ENTRY Load_FA0 LEAF_END Load_FA0 LEAF_ENTRY Load_FA0_FA1 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -1998,11 +2033,15 @@ LEAF_ENTRY Load_FA0_FA1 LEAF_END Load_FA0_FA1 LEAF_ENTRY Load_FA0_FA1_FA2 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2 +#if __riscv_flen >= 64 fld fa2, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2010,12 +2049,16 @@ ALTERNATE_ENTRY Load_FA2 LEAF_END Load_FA0_FA1_FA2 LEAF_ENTRY Load_FA0_FA1_FA2_FA3 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2023,15 +2066,21 @@ ALTERNATE_ENTRY Load_FA2_FA3 LEAF_END Load_FA0_FA1_FA2_FA3 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4 +#if __riscv_flen >= 64 fld fa4, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2039,16 +2088,22 @@ ALTERNATE_ENTRY Load_FA4 LEAF_END Load_FA0_FA1_FA2_FA3_FA4 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4_FA5 +#if __riscv_flen >= 64 fld fa4, 0(t3) fld fa5, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2056,19 +2111,27 @@ ALTERNATE_ENTRY Load_FA4_FA5 LEAF_END Load_FA0_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa4, 0(t3) fld fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA6 +#if __riscv_flen >= 64 fld fa6, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2076,20 +2139,28 @@ ALTERNATE_ENTRY Load_FA6 LEAF_END Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa4, 0(t3) fld fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA6_FA7 +#if __riscv_flen >= 64 fld fa6, 0(t3) fld fa7, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2098,7 +2169,9 @@ LEAF_END Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 // Additional Load_FA* routines starting from FA1, FA3, FA5, FA7 LEAF_ENTRY Load_FA1 +#if __riscv_flen >= 64 fld fa1, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2106,8 +2179,10 @@ LEAF_ENTRY Load_FA1 LEAF_END Load_FA1 LEAF_ENTRY Load_FA1_FA2 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2115,9 +2190,11 @@ LEAF_ENTRY Load_FA1_FA2 LEAF_END Load_FA1_FA2 LEAF_ENTRY Load_FA1_FA2_FA3 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2125,10 +2202,12 @@ LEAF_ENTRY Load_FA1_FA2_FA3 LEAF_END Load_FA1_FA2_FA3 LEAF_ENTRY Load_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) fld fa4, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2136,11 +2215,13 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4 LEAF_END Load_FA1_FA2_FA3_FA4 LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) fld fa4, 24(t3) fld fa5, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2148,12 +2229,14 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5 LEAF_END Load_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) fld fa4, 24(t3) fld fa5, 32(t3) fld fa6, 40(t3) +#endif addi t3, t3, 48 ld t4, 0(t2) addi t2, t2, 8 @@ -2161,6 +2244,7 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_END Load_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) @@ -2168,6 +2252,7 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 fld fa5, 32(t3) fld fa6, 40(t3) fld fa7, 48(t3) +#endif addi t3, t3, 56 ld t4, 0(t2) addi t2, t2, 8 @@ -2175,7 +2260,9 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_END Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Load_FA3 +#if __riscv_flen >= 64 fld fa3, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2183,8 +2270,10 @@ LEAF_ENTRY Load_FA3 LEAF_END Load_FA3 LEAF_ENTRY Load_FA3_FA4 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2192,9 +2281,11 @@ LEAF_ENTRY Load_FA3_FA4 LEAF_END Load_FA3_FA4 LEAF_ENTRY Load_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) fld fa5, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2202,10 +2293,12 @@ LEAF_ENTRY Load_FA3_FA4_FA5 LEAF_END Load_FA3_FA4_FA5 LEAF_ENTRY Load_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) fld fa5, 16(t3) fld fa6, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2213,11 +2306,13 @@ LEAF_ENTRY Load_FA3_FA4_FA5_FA6 LEAF_END Load_FA3_FA4_FA5_FA6 LEAF_ENTRY Load_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) fld fa5, 16(t3) fld fa6, 24(t3) fld fa7, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2225,7 +2320,9 @@ LEAF_ENTRY Load_FA3_FA4_FA5_FA6_FA7 LEAF_END Load_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Load_FA5 +#if __riscv_flen >= 64 fld fa5, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2233,8 +2330,10 @@ LEAF_ENTRY Load_FA5 LEAF_END Load_FA5 LEAF_ENTRY Load_FA5_FA6 +#if __riscv_flen >= 64 fld fa5, 0(t3) fld fa6, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2242,9 +2341,11 @@ LEAF_ENTRY Load_FA5_FA6 LEAF_END Load_FA5_FA6 LEAF_ENTRY Load_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa5, 0(t3) fld fa6, 8(t3) fld fa7, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2252,7 +2353,9 @@ LEAF_ENTRY Load_FA5_FA6_FA7 LEAF_END Load_FA5_FA6_FA7 LEAF_ENTRY Load_FA7 +#if __riscv_flen >= 64 fld fa7, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2260,7 +2363,9 @@ LEAF_ENTRY Load_FA7 LEAF_END Load_FA7 LEAF_ENTRY Store_FA0 +#if __riscv_flen >= 64 fsd fa0, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2268,8 +2373,10 @@ LEAF_ENTRY Store_FA0 LEAF_END Store_FA0 LEAF_ENTRY Store_FA0_FA1 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2277,11 +2384,15 @@ LEAF_ENTRY Store_FA0_FA1 LEAF_END Store_FA0_FA1 LEAF_ENTRY Store_FA0_FA1_FA2 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2 +#if __riscv_flen >= 64 fsd fa2, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2289,12 +2400,16 @@ ALTERNATE_ENTRY Store_FA2 LEAF_END Store_FA0_FA1_FA2 LEAF_ENTRY Store_FA0_FA1_FA2_FA3 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2302,15 +2417,21 @@ ALTERNATE_ENTRY Store_FA2_FA3 LEAF_END Store_FA0_FA1_FA2_FA3 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4 +#if __riscv_flen >= 64 fsd fa4, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2318,16 +2439,22 @@ ALTERNATE_ENTRY Store_FA4 LEAF_END Store_FA0_FA1_FA2_FA3_FA4 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4_FA5 +#if __riscv_flen >= 64 fsd fa4, 0(t3) fsd fa5, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2335,19 +2462,27 @@ ALTERNATE_ENTRY Store_FA4_FA5 LEAF_END Store_FA0_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa4, 0(t3) fsd fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA6 +#if __riscv_flen >= 64 fsd fa6, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2355,20 +2490,28 @@ ALTERNATE_ENTRY Store_FA6 LEAF_END Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa4, 0(t3) fsd fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA6_FA7 +#if __riscv_flen >= 64 fsd fa6, 0(t3) fsd fa7, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2377,7 +2520,9 @@ LEAF_END Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 // Additional Store_FA* routines starting from FA1, FA3, FA5, FA7 LEAF_ENTRY Store_FA1 +#if __riscv_flen >= 64 fsd fa1, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2385,8 +2530,10 @@ LEAF_ENTRY Store_FA1 LEAF_END Store_FA1 LEAF_ENTRY Store_FA1_FA2 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2394,9 +2541,11 @@ LEAF_ENTRY Store_FA1_FA2 LEAF_END Store_FA1_FA2 LEAF_ENTRY Store_FA1_FA2_FA3 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2404,10 +2553,12 @@ LEAF_ENTRY Store_FA1_FA2_FA3 LEAF_END Store_FA1_FA2_FA3 LEAF_ENTRY Store_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) fsd fa4, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2415,11 +2566,13 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4 LEAF_END Store_FA1_FA2_FA3_FA4 LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) fsd fa4, 24(t3) fsd fa5, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2427,12 +2580,14 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5 LEAF_END Store_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) fsd fa4, 24(t3) fsd fa5, 32(t3) fsd fa6, 40(t3) +#endif addi t3, t3, 48 ld t4, 0(t2) addi t2, t2, 8 @@ -2440,6 +2595,7 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_END Store_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) @@ -2447,6 +2603,7 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 fsd fa5, 32(t3) fsd fa6, 40(t3) fsd fa7, 48(t3) +#endif addi t3, t3, 56 ld t4, 0(t2) addi t2, t2, 8 @@ -2454,7 +2611,9 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_END Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Store_FA3 +#if __riscv_flen >= 64 fsd fa3, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2462,8 +2621,10 @@ LEAF_ENTRY Store_FA3 LEAF_END Store_FA3 LEAF_ENTRY Store_FA3_FA4 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2471,9 +2632,11 @@ LEAF_ENTRY Store_FA3_FA4 LEAF_END Store_FA3_FA4 LEAF_ENTRY Store_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) fsd fa5, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2481,10 +2644,12 @@ LEAF_ENTRY Store_FA3_FA4_FA5 LEAF_END Store_FA3_FA4_FA5 LEAF_ENTRY Store_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) fsd fa5, 16(t3) fsd fa6, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2492,11 +2657,13 @@ LEAF_ENTRY Store_FA3_FA4_FA5_FA6 LEAF_END Store_FA3_FA4_FA5_FA6 LEAF_ENTRY Store_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) fsd fa5, 16(t3) fsd fa6, 24(t3) fsd fa7, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2504,7 +2671,9 @@ LEAF_ENTRY Store_FA3_FA4_FA5_FA6_FA7 LEAF_END Store_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Store_FA5 +#if __riscv_flen >= 64 fsd fa5, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2512,8 +2681,10 @@ LEAF_ENTRY Store_FA5 LEAF_END Store_FA5 LEAF_ENTRY Store_FA5_FA6 +#if __riscv_flen >= 64 fsd fa5, 0(t3) fsd fa6, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2521,9 +2692,11 @@ LEAF_ENTRY Store_FA5_FA6 LEAF_END Store_FA5_FA6 LEAF_ENTRY Store_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa5, 0(t3) fsd fa6, 8(t3) fsd fa7, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2531,7 +2704,9 @@ LEAF_ENTRY Store_FA5_FA6_FA7 LEAF_END Store_FA5_FA6_FA7 LEAF_ENTRY Store_FA7 +#if __riscv_flen >= 64 fsd fa7, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 diff --git a/src/coreclr/vm/riscv64/calldescrworkerriscv64.S b/src/coreclr/vm/riscv64/calldescrworkerriscv64.S index 54725758b41b27..73e901a15e11ca 100644 --- a/src/coreclr/vm/riscv64/calldescrworkerriscv64.S +++ b/src/coreclr/vm/riscv64/calldescrworkerriscv64.S @@ -44,6 +44,7 @@ LOCAL_LABEL(donestack): ld t4, CallDescrData__pFloatArgumentRegisters(s1) beq t4, zero, LOCAL_LABEL(NoFloatingPoint) +#if __riscv_flen >= 64 fld fa0, 0(t4) fld fa1, 8(t4) fld fa2, 16(t4) @@ -52,6 +53,7 @@ LOCAL_LABEL(donestack): fld fa5, 40(t4) fld fa6, 48(t4) fld fa7, 56(t4) +#endif LOCAL_LABEL(NoFloatingPoint): // Copy [pArgumentRegisters, ..., pArgumentRegisters + 56] @@ -80,7 +82,9 @@ LOCAL_LABEL(CallDescrWorkerInternalReturnAddress): // Just save the returned registers (fa0, fa1/a0) and let CopyReturnedFpStructFromRegisters worry about placing // the fields as they were originally laid out in memory. +#if __riscv_flen >= 64 fsd fa0, CallDescrData__returnValue(s1) // fa0 is always occupied; we have at least one floating field +#endif andi a3, a3, FpStruct__BothFloat bne a3, zero, LOCAL_LABEL(SecondFieldFloatReturn) @@ -91,7 +95,9 @@ LOCAL_LABEL(CallDescrWorkerInternalReturnAddress): j LOCAL_LABEL(ReturnDone) LOCAL_LABEL(SecondFieldFloatReturn): +#if __riscv_flen >= 64 fsd fa1, (CallDescrData__returnValue + 8)(s1) +#endif j LOCAL_LABEL(ReturnDone) LOCAL_LABEL(IntReturn): diff --git a/src/coreclr/vm/riscv64/thunktemplates.S b/src/coreclr/vm/riscv64/thunktemplates.S index ce4e1d4b912025..34fae32559e42d 100644 --- a/src/coreclr/vm/riscv64/thunktemplates.S +++ b/src/coreclr/vm/riscv64/thunktemplates.S @@ -14,6 +14,9 @@ LEAF_END_MARKED StubPrecodeCode LEAF_ENTRY FixupPrecodeCode auipc t2, 0x4 ld t2, (FixupPrecodeData__Target)(t2) + // Without the C extension the tail jump is 4 bytes instead of 2, shifting + // the rest of the thunk by 2. Keep in sync with FixupPrecode::FixupCodeOffset. +#ifdef __riscv_compressed c.jr t2 fence r,rw @@ -21,6 +24,15 @@ LEAF_ENTRY FixupPrecodeCode ld t1, (FixupPrecodeData__PrecodeFixupThunk - 0xe)(t2) ld t2, (FixupPrecodeData__MethodDesc - 0xe)(t2) jr t1 +#else + jr t2 + + fence r,rw + auipc t2, 0x4 + ld t1, (FixupPrecodeData__PrecodeFixupThunk - 0x10)(t2) + ld t2, (FixupPrecodeData__MethodDesc - 0x10)(t2) + jr t1 +#endif LEAF_END_MARKED FixupPrecodeCode #ifdef FEATURE_TIERED_COMPILATION diff --git a/src/native/external/libunwind/include/libunwind-riscv.h b/src/native/external/libunwind/include/libunwind-riscv.h index 55605fe779a600..075eacd27052b8 100644 --- a/src/native/external/libunwind/include/libunwind-riscv.h +++ b/src/native/external/libunwind/include/libunwind-riscv.h @@ -68,6 +68,12 @@ typedef int64_t unw_sword_t; typedef double unw_tdep_fpreg_t; #elif __riscv_flen == 32 typedef float unw_tdep_fpreg_t; +#elif !defined(__riscv_flen) +/* Built for a target without F/D. There are no floating-point registers to + unwind, but the type is part of the public API, so keep it the width the + double-precision ABI uses and make it an integer so that the header does not + require floating point of its includer. */ +typedef uint64_t unw_tdep_fpreg_t; #else # error "Unsupported RISC-V floating-point size" #endif diff --git a/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h b/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h index b0aebc35801bff..6c752a3dc00613 100644 --- a/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h +++ b/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h @@ -162,6 +162,12 @@ dwarf_put (struct dwarf_cursor *c, dwarf_loc_t loc, unw_word_t val) static inline int dwarf_getfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t *val) { +#if !defined(__riscv_flen) + /* No F/D extension: the target has no floating-point registers, so no + floating-point location can be valid. */ + (void) c; (void) loc; (void) val; + return -UNW_EBADREG; +#else char *valp = (char *) &val; unw_word_t addr; @@ -180,11 +186,16 @@ dwarf_getfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t *val) #else # error "FIXME" #endif +#endif /* !defined(__riscv_flen) */ } static inline int dwarf_putfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t val) { +#if !defined(__riscv_flen) + (void) c; (void) loc; (void) val; + return -UNW_EBADREG; +#else char *valp = (char *) &val; unw_word_t addr; @@ -203,6 +214,7 @@ dwarf_putfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t val) #else # error "FIXME" #endif +#endif /* !defined(__riscv_flen) */ } static inline int diff --git a/src/native/external/libunwind/src/riscv/asm.h b/src/native/external/libunwind/src/riscv/asm.h index 7f7b444f931c91..2939ab7fe134c9 100644 --- a/src/native/external/libunwind/src/riscv/asm.h +++ b/src/native/external/libunwind/src/riscv/asm.h @@ -40,7 +40,8 @@ WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. */ # define SZFREG 4 # define STORE_FP fsw # define LOAD_FP flw -#else -# error "Unsupported RISC-V floating-point length" #endif +/* Without F/D there are no floating-point registers to save or restore, so + SZFREG/STORE_FP/LOAD_FP stay undefined; getcontext.S and setcontext.S + already guard their use with #ifdef STORE_FP / #ifdef LOAD_FP. */ diff --git a/src/native/minipal/cpufeatures.c b/src/native/minipal/cpufeatures.c index ed7d7612779842..f400388f0cc9fb 100644 --- a/src/native/minipal/cpufeatures.c +++ b/src/native/minipal/cpufeatures.c @@ -775,9 +775,16 @@ int minipal_getcpufeatures(void) if (syscall(__NR_riscv_hwprobe, pairs, 1, 0, NULL, 0) == 0) { - // Our baseline support is for RV64GC (see #73437) + // The hardware must implement at least the ISA the runtime was built + // for. RV64GC is the supported baseline (see #73437); a build for a + // reduced -march only assumes the extensions it was compiled with. + // (hwprobe has no F-only bit, so an F-without-D build asserts nothing.) +#if defined(__riscv_flen) && __riscv_flen >= 64 assert(pairs[0].value & RISCV_HWPROBE_IMA_FD); +#endif +#ifdef __riscv_compressed assert(pairs[0].value & RISCV_HWPROBE_IMA_C); +#endif if (pairs[0].value & RISCV_HWPROBE_EXT_ZBA) { diff --git a/src/tests/JIT/Directed/softfloat/SoftFloat.cs b/src/tests/JIT/Directed/softfloat/SoftFloat.cs new file mode 100644 index 00000000000000..e6f7f32d87c918 --- /dev/null +++ b/src/tests/JIT/Directed/softfloat/SoftFloat.cs @@ -0,0 +1,284 @@ +// Licensed to the .NET Foundation under one or more agreements. +// The .NET Foundation licenses this file to you under the MIT license. +// +// Floating-point semantics that a soft-float target (RISC-V lp64, where the +// JIT expands every FP operation into a helper call) must reproduce bit for +// bit. All expected values are exact IEEE 754 results, so the test is equally +// valid on hard-float targets: arithmetic rounding, comparison semantics with +// NaN (including the unordered branch forms), the saturating .NET conversions +// to every integer width, the checked conversions, conversions from every +// integer width, float <-> double, negation, remainder, Math intrinsics that +// become calls, values passed and returned across call boundaries in integer +// registers, and the sign extension of float bits reinterpreted as int. + +using System; +using System.Runtime.CompilerServices; +using Xunit; + +public class SoftFloat +{ + static int s_failures; + + static void Check(bool ok, string name) + { + if (!ok) + { + s_failures++; + Console.WriteLine("FAIL: " + name); + } + } + + static void CheckBits(double actual, ulong expected, string name) + => Check((ulong)BitConverter.DoubleToInt64Bits(actual) == expected, + name + ": " + BitConverter.DoubleToInt64Bits(actual).ToString("X16") + " != " + expected.ToString("X16")); + + static void CheckBits(float actual, uint expected, string name) + => Check((uint)BitConverter.SingleToInt32Bits(actual) == expected, + name + ": " + BitConverter.SingleToInt32Bits(actual).ToString("X8") + " != " + expected.ToString("X8")); + + [MethodImpl(MethodImplOptions.NoInlining)] static double D(double x) => x; + [MethodImpl(MethodImplOptions.NoInlining)] static float F(float x) => x; + [MethodImpl(MethodImplOptions.NoInlining)] static int I(int x) => x; + [MethodImpl(MethodImplOptions.NoInlining)] static long L(long x) => x; + + static void Arithmetic() + { + double a = D(0.1), b = D(0.2); + CheckBits(a + b, 0x3FD3333333333334, "0.1 + 0.2"); + CheckBits(D(1.0) / D(3.0), 0x3FD5555555555555, "1 / 3"); + CheckBits(D(100.0) / D(7.0), 0x402C924924924925, "100 / 7"); + CheckBits(D(3.7) * D(2.1), 0x401F147AE147AE15, "3.7 * 2.1"); + CheckBits(D(1e308) * D(10.0), 0x7FF0000000000000, "overflow to +inf"); + CheckBits(D(-1e308) * D(10.0), 0xFFF0000000000000, "overflow to -inf"); + CheckBits(D(double.Epsilon) * D(0.5), 0x0000000000000000, "underflow to 0"); + CheckBits(D(0.0) + D(-0.0), 0x0000000000000000, "0 + -0 = +0"); + CheckBits(D(-0.0) - D(0.0), 0x8000000000000000, "-0 - 0 = -0"); + CheckBits(D(-0.0) * D(1.0), 0x8000000000000000, "-0 * 1 = -0"); + Check(double.IsNaN(D(double.PositiveInfinity) - D(double.PositiveInfinity)), "inf - inf"); + Check(double.IsNaN(D(0.0) / D(0.0)), "0 / 0"); + CheckBits(D(1.0) / D(0.0), 0x7FF0000000000000, "1 / 0"); + CheckBits(D(-1.0) / D(0.0), 0xFFF0000000000000, "-1 / 0"); + CheckBits(D(9007199254740992.0) + D(1.0), 0x4340000000000000, "2^53 + 1 rounds to even"); + + float fa = F(0.1f), fb = F(0.2f); + CheckBits(fa + fb, 0x3E99999Au, "0.1f + 0.2f"); + CheckBits(F(1.0f) / F(3.0f), 0x3EAAAAABu, "1f / 3f"); + CheckBits(F(7.0f) / F(3.0f), 0x40155555u, "7f / 3f"); + CheckBits(F(1e10f) * F(1e10f), 0x60AD78ECu, "1e10f * 1e10f"); + CheckBits(F(float.MaxValue) * F(2.0f), 0x7F800000u, "float overflow to +inf"); + CheckBits(F(-0.0f) * F(1.0f), 0x80000000u, "-0f * 1f = -0f"); + Check(float.IsNaN(F(0.0f) / F(0.0f)), "0f / 0f"); + + // Remainder goes through the pre-existing helpers. + Check(D(5.5) % D(2.0) == 1.5, "5.5 % 2"); + Check(D(-5.5) % D(2.0) == -1.5, "-5.5 % 2"); + Check(D(5.5) % D(double.PositiveInfinity) == 5.5, "x % inf"); + Check(double.IsNaN(D(5.5) % D(0.0)), "x % 0"); + Check(F(5.5f) % F(2.0f) == 1.5f, "5.5f % 2f"); + + // The same expression twice: CSE / value numbering of the helper calls. + double c1 = a * b + a / b; + double c2 = a * b + a / b; + Check(BitConverter.DoubleToInt64Bits(c1) == BitConverter.DoubleToInt64Bits(c2), "CSE"); + } + + static void Negation() + { + CheckBits(-D(1.5), 0xBFF8000000000000, "-1.5"); + CheckBits(-D(0.0), 0x8000000000000000, "-(+0)"); + CheckBits(-D(-0.0), 0x0000000000000000, "-(-0)"); + Check(double.IsNaN(-D(double.NaN)), "-NaN"); + CheckBits(-D(double.PositiveInfinity), 0xFFF0000000000000, "-inf"); + CheckBits(-F(1.5f), 0xBFC00000u, "-1.5f"); + CheckBits(-F(0.0f), 0x80000000u, "-(+0f)"); + Check(float.IsNaN(-F(float.NaN)), "-NaNf"); + } + + static void Comparisons() + { + double nan = D(double.NaN), one = D(1.0), two = D(2.0), nz = D(-0.0), pz = D(0.0); + + Check(!(nan == nan), "NaN == NaN"); + Check(nan != nan, "NaN != NaN"); + Check(!(nan < one) && !(nan <= one) && !(nan > one) && !(nan >= one), "NaN ordered compares"); + Check(!(one < nan) && !(one <= nan) && !(one > nan) && !(one >= nan), "ordered compares with NaN rhs"); + Check(one < two && one <= two && two > one && two >= one && one <= one && one >= one, "ordered"); + Check(!(two < one) && !(two <= one) && !(one > two) && !(one >= two), "ordered false"); + Check(nz == pz && !(nz < pz) && nz <= pz && nz >= pz, "-0 == +0"); + Check(D(double.NegativeInfinity) < D(double.MinValue) && D(double.MaxValue) < D(double.PositiveInfinity), "infinities"); + + // Unordered branch forms (bge.un etc.) come from the negated conditions. + int hits = 0; + if (!(nan < one)) hits |= 1; // uge + if (!(nan <= one)) hits |= 2; // ugt + if (!(nan > one)) hits |= 4; // ule + if (!(nan >= one)) hits |= 8; // ult + if (!(nan == one)) hits |= 16; // une + Check(hits == 31, "unordered branches with NaN"); + hits = 0; + if (!(one < two)) hits |= 1; + if (!(two <= one)) hits |= 2; + if (!(one > two)) hits |= 4; + if (!(two >= one)) hits |= 8; + if (!(one == two)) hits |= 16; + Check(hits == 2 + 4 + 16, "unordered branches, ordered operands"); + + float fn = F(float.NaN), f1 = F(1.0f), f2 = F(2.0f); + Check(!(fn == fn) && fn != fn && !(fn < f1) && !(fn >= f1), "float NaN compares"); + Check(f1 < f2 && f2 >= f1 && !(f2 <= f1), "float ordered"); + Check(F(-0.0f) == F(0.0f), "-0f == +0f"); + } + + static void ToInteger() + { + // .NET semantics: NaN -> 0, saturation to the range of the destination. + Check((int)D(double.NaN) == 0, "(int)NaN"); + Check((int)D(1e10) == int.MaxValue, "(int)1e10"); + Check((int)D(-1e10) == int.MinValue, "(int)-1e10"); + Check((int)D(2147483647.9) == int.MaxValue, "(int)2147483647.9"); + Check((int)D(-2147483648.9) == int.MinValue, "(int)-2147483648.9"); + Check((int)D(-1.9) == -1 && (int)D(1.9) == 1, "(int) truncation"); + Check((int)D(double.PositiveInfinity) == int.MaxValue && (int)D(double.NegativeInfinity) == int.MinValue, "(int)inf"); + Check((uint)D(-1.0) == 0 && (uint)D(double.NaN) == 0, "(uint) negative/NaN"); + Check((uint)D(5e9) == uint.MaxValue && (uint)D(4294967295.9) == uint.MaxValue, "(uint) saturation"); + Check((uint)D(3.99) == 3 && (uint)D(4294967295.0) == uint.MaxValue, "(uint) values"); + Check((long)D(double.NaN) == 0 && (long)D(1e30) == long.MaxValue && (long)D(-1e30) == long.MinValue, "(long)"); + Check((long)D(-9223372036854775808.0) == long.MinValue && (long)D(9223372036854775807.0) == long.MaxValue, "(long) edges"); + Check((ulong)D(-1.0) == 0 && (ulong)D(1e30) == ulong.MaxValue && (ulong)D(18446744073709551615.0) == ulong.MaxValue, "(ulong)"); + Check((ulong)D(9223372036854775808.0) == 9223372036854775808UL, "(ulong)2^63"); + // Conversions to the small integer types saturate as well (.NET 11). + Check((byte)D(300.0) == 255 && (byte)D(-5.0) == 0 && (byte)D(double.NaN) == 0, "(byte)"); + Check((sbyte)D(-200.0) == -128 && (sbyte)D(200.0) == 127, "(sbyte)"); + Check((short)D(70000.0) == 32767 && (ushort)D(70000.0) == 65535 && (ushort)D(-1.0) == 0, "(short)/(ushort)"); + + Check((int)F(1e10f) == int.MaxValue && (int)F(float.NaN) == 0 && (int)F(-2.5f) == -2, "(int)float"); + Check((long)F(1e30f) == long.MaxValue && (ulong)F(-1.0f) == 0, "(long)/(ulong) float"); + Check((byte)F(300.0f) == 255, "(byte)float"); + + // Checked conversions throw for NaN and out-of-range values. + Check(Throws(() => checked((int)D(1e10))), "checked (int)1e10"); + Check(Throws(() => checked((int)D(double.NaN))), "checked (int)NaN"); + Check(Throws(() => checked((uint)D(-1.0))), "checked (uint)-1"); + Check(Throws(() => checked((long)D(1e30))), "checked (long)1e30"); + Check(Throws(() => checked((byte)D(256.0))), "checked (byte)256"); + Check(checked((int)D(-2147483648.0)) == int.MinValue && checked((int)D(2147483647.0)) == int.MaxValue, "checked (int) edges"); + Check(checked((long)F(1e18f)) == 999999984306749440L, "checked (long)1e18f"); + } + + static void FromInteger() + { + CheckBits((double)I(int.MinValue), 0xC1E0000000000000, "(double)int.MinValue"); + CheckBits((double)(uint)I(-1), 0x41EFFFFFFFE00000, "(double)uint.MaxValue"); + CheckBits((double)L(long.MaxValue), 0x43E0000000000000, "(double)long.MaxValue"); + CheckBits((double)(ulong)L(-1), 0x43F0000000000000, "(double)ulong.MaxValue"); + CheckBits((double)L(long.MinValue), 0xC3E0000000000000, "(double)long.MinValue"); + CheckBits((double)L(9007199254740993), 0x4340000000000000, "(double)(2^53+1) rounds to even"); + CheckBits((float)I(16777217), 0x4B800000u, "(float)16777217"); + CheckBits((float)L(long.MaxValue), 0x5F000000u, "(float)long.MaxValue"); + CheckBits((float)(ulong)L(-1), 0x5F800000u, "(float)ulong.MaxValue"); + CheckBits((float)(uint)I(-1), 0x4F800000u, "(float)uint.MaxValue"); + CheckBits((double)(short)I(-3), 0xC008000000000000, "(double)short"); + CheckBits((double)(byte)I(255), 0x406FE00000000000, "(double)byte"); + } + + static void FloatDouble() + { + CheckBits((double)F(0.1f), 0x3FB99999A0000000, "(double)0.1f"); + CheckBits((float)D(0.1), 0x3DCCCCCDu, "(float)0.1"); + CheckBits((float)D(1e300), 0x7F800000u, "(float)1e300"); + CheckBits((float)D(-1e300), 0xFF800000u, "(float)-1e300"); + CheckBits((float)D(1e-50), 0x00000000u, "(float)1e-50"); + CheckBits((float)D(-0.0), 0x80000000u, "(float)-0.0"); + Check(float.IsNaN((float)D(double.NaN)), "(float)NaN"); + Check(double.IsNaN((double)F(float.NaN)), "(double)NaNf"); + CheckBits((double)(float)D(16777217.0), 0x4170000000000000, "(double)(float)16777217.0"); + } + + static void Bits() + { + // A float in an integer register may carry undefined upper bits; the + // reinterpretation as int must be a proper sign-extended int. + int bits = BitConverter.SingleToInt32Bits(F(-1.5f)); + Check(bits == -1077936128, "SingleToInt32Bits(-1.5f)"); + Check(bits < 0, "float sign bit as int sign"); + Check((long)bits == -1077936128L, "float bits widened"); + Check(BitConverter.Int32BitsToSingle(bits) == -1.5f, "round trip"); + Check(BitConverter.DoubleToInt64Bits(D(-2.0)) == unchecked((long)0xC000000000000000), "DoubleToInt64Bits"); + Check(BitConverter.Int64BitsToDouble(0x3FF0000000000000) == 1.0, "Int64BitsToDouble"); + } + + struct Pair { public double A; public float B; public int C; } + + [MethodImpl(MethodImplOptions.NoInlining)] + static double Sum10(double a, double b, double c, double d, double e, double f, double g, double h, double i, double j) + => a + b + c + d + e + f + g + h + i + j; + + [MethodImpl(MethodImplOptions.NoInlining)] + static float Mixed(int a, float b, long c, double d, float e) => (float)(a + b + c + d + e); + + [MethodImpl(MethodImplOptions.NoInlining)] + static float SumF11(float a, float b, float c, float d, float e, float f, float g, float h, float i, float j, float k) + => a + b + c + d + e + f + g + h + i + j + k; + + [MethodImpl(MethodImplOptions.NoInlining)] + static double Mixed12(float a, double b, int c, float d, double e, long f, float g, double h, float i, double j, float k, double l) + => a + b + c + d + e + f + g + h + i + j + k + l; + + [MethodImpl(MethodImplOptions.NoInlining)] + static Pair MakePair(double a, float b, int c) => new Pair { A = a * 2, B = b * 2, C = c * 2 }; + + static void Calls() + { + Check(Sum10(1, 2, 3, 4, 5, 6, 7, 8, 9, 10) == 55.0, "10 double args (registers and stack)"); + Check(Mixed(1, 2.5f, 3, 4.25, 5.5f) == 16.25f, "mixed int/float args"); + Check(SumF11(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11) == 66.0f, "11 float args (registers and stack)"); + Check(Mixed12(0.5f, 1.5, 2, 3.5f, 4.5, 5, 6.5f, 7.5, 8.5f, 9.5, 10.5f, 11.5) == 71.0, "12 mixed args on registers and stack"); + Pair p = MakePair(1.5, 2.5f, 3); + Check(p.A == 3.0 && p.B == 5.0f && p.C == 6, "struct with FP fields"); + double[] arr = { 1.5, 2.5, 3.5 }; + double s = 0; + foreach (double v in arr) s += v; + Check(s == 7.5, "array of doubles"); + float[] farr = { 1.5f, -2.5f }; + Check(farr[0] + farr[1] == -1.0f, "array of floats"); + } + + static void Intrinsics() + { + CheckBits(Math.Sqrt(D(2.0)), 0x3FF6A09E667F3BCD, "Sqrt(2)"); + Check(double.IsNaN(Math.Sqrt(D(-1.0))), "Sqrt(-1)"); + CheckBits(Math.Abs(D(-0.0)), 0x0000000000000000, "Abs(-0)"); + Check(Math.Abs(D(-3.5)) == 3.5 && MathF.Abs(F(-3.5f)) == 3.5f, "Abs"); + Check(double.IsNaN(Math.Max(D(double.NaN), D(1.0))) && double.IsNaN(Math.Min(D(1.0), D(double.NaN))), "Max/Min NaN"); + Check(Math.Max(D(1.0), D(2.0)) == 2.0 && Math.Min(D(1.0), D(2.0)) == 1.0, "Max/Min"); + Check(Math.Max(I(3), I(4)) == 4 && Math.Min(L(-1), L(1)) == -1, "integer Max/Min unaffected"); + Check(Math.Floor(D(-1.5)) == -2.0 && Math.Ceiling(D(1.2)) == 2.0 && Math.Round(D(2.5)) == 2.0, "Floor/Ceiling/Round"); + Check(Math.Truncate(D(-1.7)) == -1.0, "Truncate"); + } + + static bool Throws(Func f) + { + try { _ = f(); } catch (OverflowException) { return true; } + return false; + } + + [Fact] + public static int TestEntryPoint() + { + Arithmetic(); + Negation(); + Comparisons(); + ToInteger(); + FromInteger(); + FloatDouble(); + Bits(); + Calls(); + Intrinsics(); + if (s_failures != 0) + { + Console.WriteLine(s_failures + " failure(s)"); + return 1; + } + return 100; + } +} diff --git a/src/tests/JIT/Directed/softfloat/SoftFloat.csproj b/src/tests/JIT/Directed/softfloat/SoftFloat.csproj new file mode 100644 index 00000000000000..c0a7d8b5cf2b6e --- /dev/null +++ b/src/tests/JIT/Directed/softfloat/SoftFloat.csproj @@ -0,0 +1,12 @@ + + + 1 + + + PdbOnly + True + + + + +