From 23477741097be6df2fbbaeef1de3c554fa96c733 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Fri, 4 Sep 2026 15:26:55 +0100 Subject: [PATCH 01/16] [RISC-V] Model the F, D, C and A extensions as instruction sets riscv64 already models its optional extensions - Zba, Zbb, Zbs and Zicond - as InstructionSets, but the base rv64gc extensions are assumed unconditionally, so there is no way to describe a target that does not have them and no way for the JIT to ask. Add F, D, C and A to InstructionSetDesc.txt with D implying F, and seed the rv64gc baseline from them in the VM and in the AOT drivers. C also gets the usual DOTNET_EnableRiscV64Compressed opt-out, which only stops the JIT from emitting compressed encodings. F, D and A get none: a JIT that silently lost one of them would either still need FP instructions under the lp64d ABI or make Interlocked non-atomic. A target without them is selected through the AOT compiler's instruction set instead, which checks the ABI and the execution environment. The public names are the psABI extension letters, so a reduced target is spelled the way the C toolchain spells it: --instruction-set=-a,-c,-d,-f. The R2R names are spelled RiscV64F/D/C/A. The JIT and managed enums scope their members (InstructionSet.RiscV64_F), but the R2R one does not, and it is the only one of the three that is a format contract and cannot be renamed afterwards, so a bare READYTORUN_INSTRUCTION_A would be a poor permanent name. Defaults are unchanged: every riscv64 target still gets F, D, C and A unless something explicitly removes them, so this is a no-op for the supported rv64gc baseline. Instruction set definitions are part of the JIT/EE contract, so the interface GUID has to change; the series bumps it once, in a later commit. Signed-off-by: Maxim Menshikov --- src/coreclr/inc/clrconfigvalues.h | 1 + src/coreclr/inc/corinfoinstructionset.h | 24 ++++++++++++++++ src/coreclr/inc/readytoruninstructionset.h | 4 +++ src/coreclr/jit/compiler.cpp | 26 +++++++++++++++++ src/coreclr/jit/jitconfigvalues.h | 1 + .../tools/Common/InstructionSetHelpers.cs | 10 +++++++ .../Runtime/ReadyToRunInstructionSet.cs | 4 +++ .../Runtime/ReadyToRunInstructionSetHelper.cs | 4 +++ .../JitInterface/CorInfoInstructionSet.cs | 28 +++++++++++++++++++ .../ThunkGenerator/InstructionSetDesc.txt | 10 ++++++- src/coreclr/vm/codeman.cpp | 18 ++++++++++++ 11 files changed, 129 insertions(+), 1 deletion(-) diff --git a/src/coreclr/inc/clrconfigvalues.h b/src/coreclr/inc/clrconfigvalues.h index 42c47d615168c7..1adbeb64429198 100644 --- a/src/coreclr/inc/clrconfigvalues.h +++ b/src/coreclr/inc/clrconfigvalues.h @@ -720,6 +720,7 @@ RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zba, W("EnableRiscV64 RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zbb, W("EnableRiscV64Zbb"), 1, "Allows RiscV64 Zbb hardware intrinsics to be disabled") RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zbs, W("EnableRiscV64Zbs"), 1, "Allows RiscV64 Zbs hardware intrinsics to be disabled") RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Zicond, W("EnableRiscV64Zicond"), 1, "Allows RiscV64 Zicond hardware intrinsics to be disabled") +RETAIL_CONFIG_DWORD_INFO(EXTERNAL_EnableRiscV64Compressed, W("EnableRiscV64Compressed"), 1, "Allows the RiscV64 C (compressed) instruction set to be disabled") #endif /// diff --git a/src/coreclr/inc/corinfoinstructionset.h b/src/coreclr/inc/corinfoinstructionset.h index 2f347572ee34d4..027c542a7699a1 100644 --- a/src/coreclr/inc/corinfoinstructionset.h +++ b/src/coreclr/inc/corinfoinstructionset.h @@ -65,6 +65,10 @@ enum CORINFO_InstructionSet InstructionSet_Zbb=3, InstructionSet_Zbs=4, InstructionSet_Zicond=5, + InstructionSet_F=6, + InstructionSet_D=7, + InstructionSet_C=8, + InstructionSet_A=9, #endif // TARGET_RISCV64 #ifdef TARGET_WASM InstructionSet_WasmBase=1, @@ -469,6 +473,14 @@ inline CORINFO_InstructionSetFlags EnsureInstructionSetFlagsAreValid(CORINFO_Ins resultflags.RemoveInstructionSet(InstructionSet_Zbs); if (resultflags.HasInstructionSet(InstructionSet_Zicond) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) resultflags.RemoveInstructionSet(InstructionSet_Zicond); + if (resultflags.HasInstructionSet(InstructionSet_F) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) + resultflags.RemoveInstructionSet(InstructionSet_F); + if (resultflags.HasInstructionSet(InstructionSet_D) && !resultflags.HasInstructionSet(InstructionSet_F)) + resultflags.RemoveInstructionSet(InstructionSet_D); + if (resultflags.HasInstructionSet(InstructionSet_C) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) + resultflags.RemoveInstructionSet(InstructionSet_C); + if (resultflags.HasInstructionSet(InstructionSet_A) && !resultflags.HasInstructionSet(InstructionSet_RiscV64Base)) + resultflags.RemoveInstructionSet(InstructionSet_A); #endif // TARGET_RISCV64 #ifdef TARGET_WASM if (resultflags.HasInstructionSet(InstructionSet_Vector128) && !resultflags.HasInstructionSet(InstructionSet_PackedSimd)) @@ -777,6 +789,14 @@ inline const char *InstructionSetToString(CORINFO_InstructionSet instructionSet) return "Zbs"; case InstructionSet_Zicond : return "Zicond"; + case InstructionSet_F : + return "F"; + case InstructionSet_D : + return "D"; + case InstructionSet_C : + return "C"; + case InstructionSet_A : + return "A"; #endif // TARGET_RISCV64 #ifdef TARGET_WASM case InstructionSet_WasmBase : @@ -989,6 +1009,10 @@ inline CORINFO_InstructionSet InstructionSetFromR2RInstructionSet(ReadyToRunInst case READYTORUN_INSTRUCTION_Zbb: return InstructionSet_Zbb; case READYTORUN_INSTRUCTION_Zbs: return InstructionSet_Zbs; case READYTORUN_INSTRUCTION_Zicond: return InstructionSet_Zicond; + case READYTORUN_INSTRUCTION_RiscV64F: return InstructionSet_F; + case READYTORUN_INSTRUCTION_RiscV64D: return InstructionSet_D; + case READYTORUN_INSTRUCTION_RiscV64C: return InstructionSet_C; + case READYTORUN_INSTRUCTION_RiscV64A: return InstructionSet_A; #endif // TARGET_RISCV64 #ifdef TARGET_WASM case READYTORUN_INSTRUCTION_WasmBase: return InstructionSet_WasmBase; diff --git a/src/coreclr/inc/readytoruninstructionset.h b/src/coreclr/inc/readytoruninstructionset.h index d2851e91577f1e..96121c5d4f843c 100644 --- a/src/coreclr/inc/readytoruninstructionset.h +++ b/src/coreclr/inc/readytoruninstructionset.h @@ -103,6 +103,10 @@ enum ReadyToRunInstructionSet READYTORUN_INSTRUCTION_Cssc=93, READYTORUN_INSTRUCTION_Zicond=94, READYTORUN_INSTRUCTION_Fp16=95, + READYTORUN_INSTRUCTION_RiscV64F=96, + READYTORUN_INSTRUCTION_RiscV64D=97, + READYTORUN_INSTRUCTION_RiscV64C=98, + READYTORUN_INSTRUCTION_RiscV64A=99, }; diff --git a/src/coreclr/jit/compiler.cpp b/src/coreclr/jit/compiler.cpp index b16af4edf9c91f..fdca6605168ff4 100644 --- a/src/coreclr/jit/compiler.cpp +++ b/src/coreclr/jit/compiler.cpp @@ -1941,6 +1941,20 @@ void Compiler::compSetProcessor() // Add virtual vector ISA. Vector128 is part of the required Wasm SIMD baseline. instructionSetFlags.AddInstructionSet(InstructionSet_Vector128); +#elif defined(TARGET_RISCV64) + // Ensure the required baseline ISA is supported in JIT code, even if not passed in by the VM. + instructionSetFlags.AddInstructionSet(InstructionSet_RiscV64Base); + + // C is the one base extension with a config opt-out: turning it off only stops + // the JIT from emitting compressed encodings, so it is safe to honor here however + // the flags were seeded. F, D and A have no opt-out; a target without them is + // selected through the AOT compiler's instruction set. + if (JitConfig.EnableRiscV64Compressed() == 0) + { + instructionSetFlags.RemoveInstructionSet(InstructionSet_C); + } + + instructionSetFlags = EnsureInstructionSetFlagsAreValid(instructionSetFlags); #endif // TARGET_ARM64 assert(instructionSetFlags.Equals(EnsureInstructionSetFlagsAreValid(instructionSetFlags))); @@ -6245,6 +6259,18 @@ int Compiler::compCompileAfterInit(CORINFO_MODULE_HANDLE classPtr, { instructionSetFlags.AddInstructionSet(InstructionSet_Zicond); } + + // F, D and A are part of the rv64gc baseline and cannot be turned off here: a target + // without them is selected through the AOT compiler's instruction set, which also + // checks that the ABI and the execution environment allow it. + instructionSetFlags.AddInstructionSet(InstructionSet_F); + instructionSetFlags.AddInstructionSet(InstructionSet_D); + instructionSetFlags.AddInstructionSet(InstructionSet_A); + + if (JitConfig.EnableRiscV64Compressed() != 0) + { + instructionSetFlags.AddInstructionSet(InstructionSet_C); + } #endif // These calls are important and explicitly ordered to ensure that the flags are correct in diff --git a/src/coreclr/jit/jitconfigvalues.h b/src/coreclr/jit/jitconfigvalues.h index fa3a8ee0b99767..5becf2e5d92bc3 100644 --- a/src/coreclr/jit/jitconfigvalues.h +++ b/src/coreclr/jit/jitconfigvalues.h @@ -456,6 +456,7 @@ RELEASE_CONFIG_INTEGER(EnableRiscV64Zba, "EnableRiscV64Zba", RELEASE_CONFIG_INTEGER(EnableRiscV64Zbb, "EnableRiscV64Zbb", 1) // Allows RiscV64 Zbb hardware intrinsics to be disabled RELEASE_CONFIG_INTEGER(EnableRiscV64Zbs, "EnableRiscV64Zbs", 1) // Allows RiscV64 Zbs hardware intrinsics to be disabled RELEASE_CONFIG_INTEGER(EnableRiscV64Zicond, "EnableRiscV64Zicond", 1) // Allows RiscV64 Zicond hardware intrinsics to be disabled +RELEASE_CONFIG_INTEGER(EnableRiscV64Compressed, "EnableRiscV64Compressed", 1) // Allows RiscV64 compressed (C) instruction emission to be disabled #endif RELEASE_CONFIG_INTEGER(EnableEmbeddedBroadcast, "EnableEmbeddedBroadcast", 1) // Allows embedded broadcasts to be disabled diff --git a/src/coreclr/tools/Common/InstructionSetHelpers.cs b/src/coreclr/tools/Common/InstructionSetHelpers.cs index ed7f1649449091..fd256c194bb01b 100644 --- a/src/coreclr/tools/Common/InstructionSetHelpers.cs +++ b/src/coreclr/tools/Common/InstructionSetHelpers.cs @@ -94,6 +94,16 @@ public static InstructionSetSupport ConfigureInstructionSetSupport(string instru instructionSetSupportBuilder.AddSupportedInstructionSet("base"); instructionSetSupportBuilder.AddSupportedInstructionSet("simd128"); } + else if (targetArchitecture == TargetArchitecture.RiscV64) + { + // The rv64gc baseline: D implies F, so "d", "c" and "a" cover + // the G+C extensions. Reduced-ISA targets (e.g. zkVM guests) + // opt out with --instruction-set=-a,-c,-d,-f. + instructionSetSupportBuilder.AddSupportedInstructionSet("base"); + instructionSetSupportBuilder.AddSupportedInstructionSet("d"); + instructionSetSupportBuilder.AddSupportedInstructionSet("c"); + instructionSetSupportBuilder.AddSupportedInstructionSet("a"); + } bool throttleAvx512 = false; diff --git a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs index 4c5574d6870c32..efb7302babd894 100644 --- a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs +++ b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSet.cs @@ -106,5 +106,9 @@ public enum ReadyToRunInstructionSet Cssc = 93, Zicond = 94, Fp16 = 95, + RiscV64F = 96, + RiscV64D = 97, + RiscV64C = 98, + RiscV64A = 99, } } diff --git a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs index 632e3053189a72..28d370e346280c 100644 --- a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs +++ b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunInstructionSetHelper.cs @@ -77,6 +77,10 @@ public static class ReadyToRunInstructionSetHelper case InstructionSet.RiscV64_Zbb: return ReadyToRunInstructionSet.Zbb; case InstructionSet.RiscV64_Zbs: return ReadyToRunInstructionSet.Zbs; case InstructionSet.RiscV64_Zicond: return ReadyToRunInstructionSet.Zicond; + case InstructionSet.RiscV64_F: return ReadyToRunInstructionSet.RiscV64F; + case InstructionSet.RiscV64_D: return ReadyToRunInstructionSet.RiscV64D; + case InstructionSet.RiscV64_C: return ReadyToRunInstructionSet.RiscV64C; + case InstructionSet.RiscV64_A: return ReadyToRunInstructionSet.RiscV64A; default: throw new Exception("Unknown instruction set"); } diff --git a/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs b/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs index 870544e2156425..4feeb6db799930 100644 --- a/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs +++ b/src/coreclr/tools/Common/JitInterface/CorInfoInstructionSet.cs @@ -63,6 +63,10 @@ public enum InstructionSet RiscV64_Zbb = InstructionSet_RiscV64.Zbb, RiscV64_Zbs = InstructionSet_RiscV64.Zbs, RiscV64_Zicond = InstructionSet_RiscV64.Zicond, + RiscV64_F = InstructionSet_RiscV64.F, + RiscV64_D = InstructionSet_RiscV64.D, + RiscV64_C = InstructionSet_RiscV64.C, + RiscV64_A = InstructionSet_RiscV64.A, Wasm32_WasmBase = InstructionSet_Wasm32.WasmBase, Wasm32_PackedSimd = InstructionSet_Wasm32.PackedSimd, Wasm32_Vector128 = InstructionSet_Wasm32.Vector128, @@ -215,6 +219,10 @@ public enum InstructionSet_RiscV64 Zbb = 3, Zbs = 4, Zicond = 5, + F = 6, + D = 7, + C = 8, + A = 9, } public enum InstructionSet_Wasm32 @@ -615,6 +623,14 @@ public static InstructionSetFlags ExpandInstructionSetByImplicationHelper(Target resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); if (resultflags.HasInstructionSet(InstructionSet.RiscV64_Zicond)) resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_F)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_D)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_F); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_C)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_A)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_RiscV64Base); break; case TargetArchitecture.Wasm32: @@ -926,6 +942,14 @@ private static InstructionSetFlags ExpandInstructionSetByReverseImplicationHelpe resultflags.AddInstructionSet(InstructionSet.RiscV64_Zbs); if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) resultflags.AddInstructionSet(InstructionSet.RiscV64_Zicond); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_F); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_F)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_D); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_C); + if (resultflags.HasInstructionSet(InstructionSet.RiscV64_RiscV64Base)) + resultflags.AddInstructionSet(InstructionSet.RiscV64_A); break; case TargetArchitecture.Wasm32: @@ -1188,6 +1212,10 @@ public static IEnumerable ArchitectureToValidInstructionSets yield return new InstructionSetInfo("zbb", "", InstructionSet.RiscV64_Zbb, true); yield return new InstructionSetInfo("zbs", "", InstructionSet.RiscV64_Zbs, true); yield return new InstructionSetInfo("zicond", "", InstructionSet.RiscV64_Zicond, true); + yield return new InstructionSetInfo("f", "", InstructionSet.RiscV64_F, true); + yield return new InstructionSetInfo("d", "", InstructionSet.RiscV64_D, true); + yield return new InstructionSetInfo("c", "", InstructionSet.RiscV64_C, true); + yield return new InstructionSetInfo("a", "", InstructionSet.RiscV64_A, true); break; case TargetArchitecture.Wasm32: diff --git a/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt b/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt index a70cd7d9991687..dd252841739d62 100644 --- a/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt +++ b/src/coreclr/tools/Common/JitInterface/ThunkGenerator/InstructionSetDesc.txt @@ -26,7 +26,7 @@ ; DO NOT CHANGE R2R NUMERIC VALUES OF THE EXISTING SETS. Changing R2R numeric values definitions would be R2R format breaking change. ; The ISA definitions should also be mapped to `hwintrinsicIsaRangeArray` in hwintrinsic.cpp. -; NEXT_AVAILABLE_R2R_BIT = 96 +; NEXT_AVAILABLE_R2R_BIT = 100 ; Definition of X86 instruction sets definearch ,X86 ,32Bit ,X64, X64, X86 @@ -287,11 +287,19 @@ instructionset ,RiscV64 , ,Zba ,57 ,Zba ,zba instructionset ,RiscV64 , ,Zbb ,58 ,Zbb ,zbb instructionset ,RiscV64 , ,Zbs ,84 ,Zbs ,zbs instructionset ,RiscV64 , ,Zicond ,94 ,Zicond ,zicond +instructionset ,RiscV64 , ,RiscV64F ,96 ,F ,f +instructionset ,RiscV64 , ,RiscV64D ,97 ,D ,d +instructionset ,RiscV64 , ,RiscV64C ,98 ,C ,c +instructionset ,RiscV64 , ,RiscV64A ,99 ,A ,a implication ,RiscV64 ,Zbb ,RiscV64Base implication ,RiscV64 ,Zba ,RiscV64Base implication ,RiscV64 ,Zbs ,RiscV64Base implication ,RiscV64 ,Zicond ,RiscV64Base +implication ,RiscV64 ,F ,RiscV64Base +implication ,RiscV64 ,D ,F +implication ,RiscV64 ,C ,RiscV64Base +implication ,RiscV64 ,A ,RiscV64Base ; ,name and aliases ,archs ,lower baselines included by implication instructionsetgroup ,x86-64-v2 ,X64 X86 ,base diff --git a/src/coreclr/vm/codeman.cpp b/src/coreclr/vm/codeman.cpp index 4e4aa6f787a0b5..335a38f863306b 100644 --- a/src/coreclr/vm/codeman.cpp +++ b/src/coreclr/vm/codeman.cpp @@ -1789,6 +1789,24 @@ void EEJitManager::SetCpuInfo() if (g_pConfig->EnableHWIntrinsic()) { CPUCompileFlags.Set(InstructionSet_RiscV64Base); + + // F/D/C/A belong to the rv64gc baseline and are not reported through + // hwprobe; the floor is the ISA this runtime itself was built for. +#if defined(__riscv_flen) && __riscv_flen >= 32 + CPUCompileFlags.Set(InstructionSet_F); +#endif +#if defined(__riscv_flen) && __riscv_flen >= 64 + CPUCompileFlags.Set(InstructionSet_D); +#endif +#ifdef __riscv_compressed + if (CLRConfig::GetConfigValue(CLRConfig::EXTERNAL_EnableRiscV64Compressed)) + { + CPUCompileFlags.Set(InstructionSet_C); + } +#endif +#ifdef __riscv_atomic + CPUCompileFlags.Set(InstructionSet_A); +#endif } if (((cpuFeatures & RiscV64IntrinsicConstants_Zba) != 0) && CLRConfig::GetConfigValue(CLRConfig::EXTERNAL_EnableRiscV64Zba)) From 615c8a7194f47a60c3f031c939ea9df644e63fba Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Fri, 4 Sep 2026 14:06:14 +0100 Subject: [PATCH 02/16] [RISC-V] Gate FP and compressed emission on the F and C instruction sets With F, D, C and A described as instruction sets, let the riscv64 backend ask before emitting an encoding that needs one: * tryEmitCompressedIns_R_R_R - the only place the JIT emits a compressed encoding - returns false unless C is available. * A Checked JIT asserts in emitOutput_Instr that no F/D-extension major opcode is emitted unless F is available, so an unexpanded floating point node fails on the method that produced it rather than at run time on the target. The assert needs compIsaSupportedDebugOnly, which was a no-op returning false outside xarch and arm64; enable the real check for riscv64 too. Both are no-ops on the rv64gc baseline, where F, D, C and A are always present. Signed-off-by: Maxim Menshikov --- src/coreclr/jit/compiler.h | 2 +- src/coreclr/jit/emitriscv64.cpp | 30 ++++++++++++++++++++++++++++++ 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/src/coreclr/jit/compiler.h b/src/coreclr/jit/compiler.h index c09c3f8cb8b4b9..ec2b6d751e0584 100644 --- a/src/coreclr/jit/compiler.h +++ b/src/coreclr/jit/compiler.h @@ -10960,7 +10960,7 @@ class Compiler // support/nonsupport for an instruction set bool compIsaSupportedDebugOnly(CORINFO_InstructionSet isa) const { -#if defined(TARGET_XARCH) || defined(TARGET_ARM64) +#if defined(TARGET_XARCH) || defined(TARGET_ARM64) || defined(TARGET_RISCV64) return opts.compSupportsISA.HasInstructionSet(isa); #else return false; diff --git a/src/coreclr/jit/emitriscv64.cpp b/src/coreclr/jit/emitriscv64.cpp index 9dcd3a29d7aaab..02c4ddb5f41ac3 100644 --- a/src/coreclr/jit/emitriscv64.cpp +++ b/src/coreclr/jit/emitriscv64.cpp @@ -1012,6 +1012,13 @@ void emitter::emitIns_R_R_R( bool emitter::tryEmitCompressedIns_R_R_R( instruction ins, emitAttr attr, regNumber rd, regNumber rs1, regNumber rs2, insOpts opt) { + // Targets without the C extension (e.g. a zkVM guest on rv64im) must never + // receive a compressed encoding. This is the only place the JIT emits one. + if (!m_compiler->compOpportunisticallyDependsOn(InstructionSet_C)) + { + return false; + } + // TODO-RISCV64-RVC: Disable this early return once compresed instructions are allowed in prolog / epilog if (emitGeneratingPrologOrFuncletProlog() || emitGeneratingEpilogOrFuncletEpilog()) { @@ -2191,6 +2198,29 @@ unsigned emitter::emitOutput_Instr(BYTE* dst, code_t code) const { assert(dst != nullptr); static_assert(sizeof(code_t) == 4, "code_t must be 4 bytes"); +#ifdef DEBUG + // On rv64 these major opcodes decode exclusively to F/D-extension + // instructions, so a no-F target must never emit one. The compressed FP + // forms are unreachable once C is gated off; checked for completeness. + switch (GetMajorOpcode(code)) + { + case MajorOpcode::LoadFp: + case MajorOpcode::StoreFp: + case MajorOpcode::MAdd: + case MajorOpcode::MSub: + case MajorOpcode::NmSub: + case MajorOpcode::NmAdd: + case MajorOpcode::OpFp: + case MajorOpcode::Fld: + case MajorOpcode::Fsd: + case MajorOpcode::FldSp: + case MajorOpcode::FsdSp: + assert(m_compiler->compIsaSupportedDebugOnly(InstructionSet_F)); + break; + default: + break; + } +#endif // DEBUG unsigned codeSize = Is32BitInstruction((WORD)code) ? 4 : 2; assert((codeSize == 4) || ((code >> 16) == 0)); memcpy(dst + writeableOffset, &code, codeSize); From d645e95223e69f288536edd1cf3735dfd67369bc Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Tue, 8 Sep 2026 23:19:24 +0100 Subject: [PATCH 03/16] [RISC-V] Make the native build's ISA string and ABI configurable The native runtime is compiled -march=rv64gc -mabi=lp64d unconditionally, so a riscv64 target whose sysroot is built for another ABI cannot be built at all: the objects claim lp64d, the sysroot CRT claims something else, and lld refuses to reconcile the two. The only way out today is to append overriding flags after these, which works by accident of ordering and hides what the build targets. Turn the two into cache variables with the current values as defaults, so an unconfigured build is byte-identical, and a target that needs a different ABI states it once: cmake -DCLR_CMAKE_RISCV64_MABI=lp64 -DCLR_CMAKE_RISCV64_MARCH=rv64im ... The float-ABI field of e_flags follows -mabi rather than -march, so setting the ABI here is what makes the whole native build agree with its sysroot. The flags are passed at link time as well, so the driver selects the matching CRT and builtins. This mirrors how armel selects -mfloat-abi=softfp a few lines above. Signed-off-by: Maxim Menshikov --- eng/native/configurecompiler.cmake | 35 ++++++++++++++++++++++++++++-- 1 file changed, 33 insertions(+), 2 deletions(-) diff --git a/eng/native/configurecompiler.cmake b/eng/native/configurecompiler.cmake index b24f43e7d8cc42..a326d49a21675b 100644 --- a/eng/native/configurecompiler.cmake +++ b/eng/native/configurecompiler.cmake @@ -925,8 +925,39 @@ if(CLR_CMAKE_HOST_UNIX_ARMV6) endif(CLR_CMAKE_HOST_UNIX_ARMV6) if(CLR_CMAKE_HOST_UNIX_RISCV64) - add_compile_options(-march=rv64gc) - add_compile_options(-mabi=lp64d) + # The ISA string and the ABI the native runtime is built with. The defaults are + # the rv64gc/lp64d baseline, so an unconfigured build is unchanged. A target + # whose sysroot is built for a different ABI - a soft-float lp64 userspace, for + # instance - selects it here, the way armel selects -mfloat-abi=softfp above, + # instead of overriding the flags further down the command line. + # + # The float-ABI field of e_flags follows -mabi (not -march), so setting the ABI + # once here is what makes the whole native build agree with the sysroot; the + # linker rejects a mix, and it is also passed at link time so that the driver + # selects the matching CRT and builtins. + set(CLR_CMAKE_RISCV64_MARCH "rv64gc" CACHE STRING "RISC-V ISA string for the native runtime build") + set(CLR_CMAKE_RISCV64_MABI "lp64d" CACHE STRING "RISC-V ABI for the native runtime build") + + # Also settable from the environment, the way CLR_CC, ROOTFS_DIR and TOOLCHAIN + # already are: a build driven through a superproject cannot always reach the + # cmake command line of an individual repository. + if(DEFINED ENV{CLR_CMAKE_RISCV64_MARCH}) + set(CLR_CMAKE_RISCV64_MARCH "$ENV{CLR_CMAKE_RISCV64_MARCH}") + endif() + if(DEFINED ENV{CLR_CMAKE_RISCV64_MABI}) + set(CLR_CMAKE_RISCV64_MABI "$ENV{CLR_CMAKE_RISCV64_MABI}") + endif() + + # Without the A extension every __atomic_* call is "not lock-free" and clang + # warns; the runtime builds with -Werror, which would make that fatal. + if(NOT CLR_CMAKE_RISCV64_MARCH MATCHES "a") + add_compile_options(-Wno-atomic-alignment) + endif() + + add_compile_options(-march=${CLR_CMAKE_RISCV64_MARCH}) + add_compile_options(-mabi=${CLR_CMAKE_RISCV64_MABI}) + add_link_options(-march=${CLR_CMAKE_RISCV64_MARCH}) + add_link_options(-mabi=${CLR_CMAKE_RISCV64_MABI}) endif(CLR_CMAKE_HOST_UNIX_RISCV64) if(CLR_CMAKE_HOST_UNIX_X86) From c207f6c0c7a9cdcca1a26183a96b33ff5b4fa92f Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Tue, 8 Sep 2026 23:39:42 +0100 Subject: [PATCH 04/16] [RISC-V] Apply the ISA string and ABI to the cross toolchain probes configurecompiler.cmake settles the RISC-V ISA and ABI for the project, but cmake compiles and links its own probes at project() time, before any of the project's options exist. On a sysroot built for an ABI other than the compiler default those probes fail to link - lld does not reconcile float ABIs - and cmake reports the compiler as broken before the build starts. Apply the same two variables from the cross toolchain file, which is in effect for the probes as well, and list them in CMAKE_TRY_COMPILE_PLATFORM_VARIABLES so they survive nested try_compile passes. Both are unset by default, so the toolchain's own defaults stay in place and nothing changes for existing targets. This follows what the arm/armel branch above already does for -mfpu and -mfloat-abi. Signed-off-by: Maxim Menshikov --- eng/common/cross/toolchain.cmake | 45 ++++++++++++++++++++++++++++++++ 1 file changed, 45 insertions(+) diff --git a/eng/common/cross/toolchain.cmake b/eng/common/cross/toolchain.cmake index 70b71395e3ba72..4a4ef49aa930cf 100644 --- a/eng/common/cross/toolchain.cmake +++ b/eng/common/cross/toolchain.cmake @@ -333,6 +333,51 @@ if(TARGET_ARCH_NAME MATCHES "^(arm|armel)$") if(TARGET_ARCH_NAME STREQUAL "armel") add_compile_options(-mfloat-abi=softfp) endif() +elseif(TARGET_ARCH_NAME STREQUAL "riscv64") + # The ISA string and the ABI have to be settled here rather than in + # configurecompiler.cmake alone: cmake compiles and links its own probes at + # project() time, and a probe built for a different ABI than the sysroot fails + # to link, so the compiler is reported as broken before the build starts. + # + # Both are unset by default, which leaves the toolchain's own defaults in + # place. They may also come from the environment, the way CROSS_ROOTFS, + # TARGET_BUILD_ARCH and TOOLCHAIN already do, so that a build driven through a + # superproject can select them without reaching the cmake command line. + if(DEFINED ENV{CLR_CMAKE_RISCV64_MARCH}) + set(CLR_CMAKE_RISCV64_MARCH "$ENV{CLR_CMAKE_RISCV64_MARCH}") + endif() + if(DEFINED ENV{CLR_CMAKE_RISCV64_MABI}) + set(CLR_CMAKE_RISCV64_MABI "$ENV{CLR_CMAKE_RISCV64_MABI}") + endif() + + set(_riscv64_isa_flags "") + if(DEFINED CLR_CMAKE_RISCV64_MARCH) + string(APPEND _riscv64_isa_flags " -march=${CLR_CMAKE_RISCV64_MARCH}") + endif() + if(DEFINED CLR_CMAKE_RISCV64_MABI) + string(APPEND _riscv64_isa_flags " -mabi=${CLR_CMAKE_RISCV64_MABI}") + endif() + + if(NOT _riscv64_isa_flags STREQUAL "") + # *_FLAGS_INIT rather than add_compile_options: a try_compile runs as its own + # project and does not inherit directory properties, so the probe would still + # be built for the compiler's default ABI and fail against the sysroot. + string(APPEND CMAKE_C_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_CXX_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_ASM_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_EXE_LINKER_FLAGS_INIT "${_riscv64_isa_flags}") + string(APPEND CMAKE_SHARED_LINKER_FLAGS_INIT "${_riscv64_isa_flags}") + endif() + + # Without the A extension the compiler lowers C/C++ atomics to __atomic_* + # calls instead of emitting lr/sc, and those live in libatomic. Not every link + # in the tree passes -latomic on its own. + if(DEFINED CLR_CMAKE_RISCV64_MARCH AND NOT CLR_CMAKE_RISCV64_MARCH MATCHES "a") + add_toolchain_linker_flag("-latomic") + endif() + + # persist variables across multiple try_compile passes + list(APPEND CMAKE_TRY_COMPILE_PLATFORM_VARIABLES CLR_CMAKE_RISCV64_MARCH CLR_CMAKE_RISCV64_MABI) elseif(TARGET_ARCH_NAME STREQUAL "s390x") add_compile_options("--target=${TOOLCHAIN}") elseif(TARGET_ARCH_NAME STREQUAL "x86") From d8b99086b0d28c35e2f6a2e5d0d16554d3c3121c Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 9 Sep 2026 00:01:40 +0100 Subject: [PATCH 05/16] [RISC-V] Build libunwind for targets without the F/D extensions The riscv64 port refuses to compile when __riscv_flen is not 32 or 64, so a target built for the base integer ISA cannot build the PAL's unwinder at all: libunwind-riscv.h has no unw_tdep_fpreg_t, asm.h has no load/store width, and dwarf_getfp/dwarf_putfp hit an #error because __riscv_xlen != __riscv_flen. There is nothing to unwind in that configuration - the target has no floating-point registers - so the three cases are answered rather than made to work: * unw_tdep_fpreg_t is part of the public API, so it stays, as an integer of the width the double-precision ABI uses. * asm.h leaves SZFREG/STORE_FP/LOAD_FP undefined; getcontext.S and setcontext.S already guard every use with #ifdef, so they save and restore the integer registers alone. * dwarf_getfp and dwarf_putfp return -UNW_EBADREG: with no floating-point registers, no floating-point location is valid. Targets that have F or D are unaffected - the existing branches are untouched and still the only ones taken. Signed-off-by: Maxim Menshikov --- .../external/libunwind/include/libunwind-riscv.h | 6 ++++++ .../libunwind/include/tdep-riscv/libunwind_i.h | 12 ++++++++++++ src/native/external/libunwind/src/riscv/asm.h | 5 +++-- 3 files changed, 21 insertions(+), 2 deletions(-) diff --git a/src/native/external/libunwind/include/libunwind-riscv.h b/src/native/external/libunwind/include/libunwind-riscv.h index 55605fe779a600..075eacd27052b8 100644 --- a/src/native/external/libunwind/include/libunwind-riscv.h +++ b/src/native/external/libunwind/include/libunwind-riscv.h @@ -68,6 +68,12 @@ typedef int64_t unw_sword_t; typedef double unw_tdep_fpreg_t; #elif __riscv_flen == 32 typedef float unw_tdep_fpreg_t; +#elif !defined(__riscv_flen) +/* Built for a target without F/D. There are no floating-point registers to + unwind, but the type is part of the public API, so keep it the width the + double-precision ABI uses and make it an integer so that the header does not + require floating point of its includer. */ +typedef uint64_t unw_tdep_fpreg_t; #else # error "Unsupported RISC-V floating-point size" #endif diff --git a/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h b/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h index b0aebc35801bff..6c752a3dc00613 100644 --- a/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h +++ b/src/native/external/libunwind/include/tdep-riscv/libunwind_i.h @@ -162,6 +162,12 @@ dwarf_put (struct dwarf_cursor *c, dwarf_loc_t loc, unw_word_t val) static inline int dwarf_getfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t *val) { +#if !defined(__riscv_flen) + /* No F/D extension: the target has no floating-point registers, so no + floating-point location can be valid. */ + (void) c; (void) loc; (void) val; + return -UNW_EBADREG; +#else char *valp = (char *) &val; unw_word_t addr; @@ -180,11 +186,16 @@ dwarf_getfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t *val) #else # error "FIXME" #endif +#endif /* !defined(__riscv_flen) */ } static inline int dwarf_putfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t val) { +#if !defined(__riscv_flen) + (void) c; (void) loc; (void) val; + return -UNW_EBADREG; +#else char *valp = (char *) &val; unw_word_t addr; @@ -203,6 +214,7 @@ dwarf_putfp (struct dwarf_cursor *c, dwarf_loc_t loc, unw_fpreg_t val) #else # error "FIXME" #endif +#endif /* !defined(__riscv_flen) */ } static inline int diff --git a/src/native/external/libunwind/src/riscv/asm.h b/src/native/external/libunwind/src/riscv/asm.h index 7f7b444f931c91..2939ab7fe134c9 100644 --- a/src/native/external/libunwind/src/riscv/asm.h +++ b/src/native/external/libunwind/src/riscv/asm.h @@ -40,7 +40,8 @@ WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. */ # define SZFREG 4 # define STORE_FP fsw # define LOAD_FP flw -#else -# error "Unsupported RISC-V floating-point length" #endif +/* Without F/D there are no floating-point registers to save or restore, so + SZFREG/STORE_FP/LOAD_FP stay undefined; getcontext.S and setcontext.S + already guard their use with #ifdef STORE_FP / #ifdef LOAD_FP. */ From 01310ba0e7d3beba7fee5d1873cdf34b87278e21 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 9 Sep 2026 11:26:54 +0100 Subject: [PATCH 06/16] [RISC-V] Guard FP and atomic instructions in the assembly sources The hand-written riscv64 assembly uses fld/fsd and the A-extension amo* unconditionally, so a runtime built for a target without F/D or A does not assemble at all - the failure is in the assembler, before any of the ISA modelling in the compiler can help. Guard those sequences on the __riscv_flen and __riscv_atomic predefined macros, which are exactly what -march sets, and provide the lr/sc-free fallback for the atomics. A build that has the extensions is unchanged: the guards are true and the same instructions are emitted. The .S sources are the one place where the instruction set cannot be a compiler concept - the assembler is the consumer - so they switch on the predefined macros rather than on the InstructionSet model used elsewhere. Signed-off-by: Maxim Menshikov --- .../debug/di/riscv64/floatconversion.S | 2 + .../Runtime/riscv64/ExceptionHandling.S | 44 +++++ .../nativeaot/Runtime/riscv64/GcProbe.S | 4 + .../Runtime/riscv64/UniversalTransition.S | 6 + src/coreclr/pal/inc/unixasmmacrosriscv64.inc | 10 + src/coreclr/pal/src/arch/riscv64/context2.S | 8 + src/coreclr/runtime/riscv64/WriteBarriers.S | 13 ++ src/coreclr/vm/precode.h | 6 + src/coreclr/vm/riscv64/asmhelpers.S | 175 ++++++++++++++++++ .../vm/riscv64/calldescrworkerriscv64.S | 6 + src/coreclr/vm/riscv64/thunktemplates.S | 12 ++ 11 files changed, 286 insertions(+) diff --git a/src/coreclr/debug/di/riscv64/floatconversion.S b/src/coreclr/debug/di/riscv64/floatconversion.S index 138db0bc9dd243..1b8e02957d387e 100644 --- a/src/coreclr/debug/di/riscv64/floatconversion.S +++ b/src/coreclr/debug/di/riscv64/floatconversion.S @@ -7,6 +7,8 @@ // input: (in A0) the address of the ULONGLONG to be converted to a double // output: the double corresponding to the ULONGLONG input value LEAF_ENTRY FPFillR8, .TEXT +#if __riscv_flen >= 64 fld fa0, 0(a0) +#endif ret LEAF_END FPFillR8, .TEXT diff --git a/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S b/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S index 8cbe1b6a276982..21b1ac1b2e22a7 100644 --- a/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S +++ b/src/coreclr/nativeaot/Runtime/riscv64/ExceptionHandling.S @@ -31,6 +31,7 @@ .endif // Safely using available registers for floating-point saves +#if __riscv_flen >= 64 fsd fs0, 0x10(sp) fsd fs1, 0x18(sp) fsd fs2, 0x20(sp) @@ -43,6 +44,7 @@ fsd fs9, 0x58(sp) fsd fs10, 0x60(sp) fsd fs11, 0x68(sp) +#endif PROLOG_SAVE_REG_PAIR_INDEXED fp, ra, 0x78 @@ -140,6 +142,7 @@ // Load FP preserved registers // addi t3, \regdisplayReg, OFFSETOF__REGDISPLAY__F // Base address of floating-point registers +#if __riscv_flen >= 64 fld fs0, 0x40(t3) // Load fs0 fld fs1, 0x48(t3) // Load fs1 fld fs2, 0x90(t3) // Load fs2 @@ -152,6 +155,7 @@ fld fs9, 0xc8(t3) // Load fs9 fld fs10, 0xd0(t3) // Load fs10 fld fs11, 0xd8(t3) // Load fs11 +#endif .endm @@ -188,6 +192,7 @@ // Save floating-point registers addi t3, \regdisplayReg, OFFSETOF__REGDISPLAY__F +#if __riscv_flen >= 64 fsd fs0, 0x40(t3) fsd fs1, 0x48(t3) fsd fs2, 0x90(t3) @@ -200,6 +205,7 @@ fsd fs9, 0xc8(t3) fsd fs10, 0xd0(t3) fsd fs11, 0xd8(t3) +#endif .endm @@ -474,6 +480,7 @@ LOCAL_LABEL(NotHijacked): ALLOC_CALL_FUNCLET_FRAME 0x90 // Save floating-point registers +#if __riscv_flen >= 64 fsd fs0, 0x00(sp) fsd fs1, 0x08(sp) fsd fs2, 0x10(sp) @@ -486,6 +493,7 @@ LOCAL_LABEL(NotHijacked): fsd fs9, 0x48(sp) fsd fs10, 0x50(sp) fsd fs11, 0x58(sp) +#endif // Save integer registers sd a0, 0x60(sp) // Save a0 to a3 @@ -511,7 +519,14 @@ LOCAL_LABEL(NotHijacked): addi t3, a5, OFFSETOF__Thread__m_ThreadStateFlags addiw a6, zero, -17 // Mask value (0xFFFFFFEF) +#ifdef __riscv_atomic amoand.w a4, a6, (t3) +#else + // No A extension: non-atomic read-modify-write. + lw a4, (t3) + and a4, a4, a6 + sw a4, (t3) +#endif // Set preserved regs to the values expected by the funclet RESTORE_PRESERVED_REGISTERS a2 @@ -593,6 +608,7 @@ LOCAL_LABEL(DonePopping): ALLOC_CALL_FUNCLET_FRAME 0x80 // Save floating-point registers +#if __riscv_flen >= 64 fsd fs0, 0x00(sp) fsd fs1, 0x08(sp) fsd fs2, 0x10(sp) @@ -605,6 +621,7 @@ LOCAL_LABEL(DonePopping): fsd fs9, 0x48(sp) fsd fs10, 0x50(sp) fsd fs11, 0x58(sp) +#endif // Save integer registers sd a0, 0x60(sp) // Save a0 to 0x60 @@ -623,7 +640,14 @@ LOCAL_LABEL(DonePopping): // Set the DoNotTriggerGc flag addi t3, a2, OFFSETOF__Thread__m_ThreadStateFlags addiw a3, zero, -17 // Mask value (0xFFFFFFEF) +#ifdef __riscv_atomic amoand.w a4, a3, (t3) +#else + // No A extension: non-atomic read-modify-write. + lw a4, (t3) + and a4, a4, a3 + sw a4, (t3) +#endif // Restore preserved registers RESTORE_PRESERVED_REGISTERS a1 @@ -646,9 +670,17 @@ LOCAL_LABEL(DonePopping): addi t3, a2, OFFSETOF__Thread__m_ThreadStateFlags addiw a3, zero, 16 // Mask value (0x10) +#ifdef __riscv_atomic amoor.w a1, a3, (t3) +#else + // No A extension: non-atomic read-modify-write. + lw a1, (t3) + or a1, a1, a3 + sw a1, (t3) +#endif // Restore floating-point registers +#if __riscv_flen >= 64 fld fs0, 0x00(sp) fld fs1, 0x08(sp) fld fs2, 0x10(sp) @@ -661,6 +693,7 @@ LOCAL_LABEL(DonePopping): fld fs9, 0x48(sp) fld fs10, 0x50(sp) fld fs11, 0x58(sp) +#endif // Free call funclet frame FREE_CALL_FUNCLET_FRAME 0x80 @@ -685,6 +718,7 @@ LOCAL_LABEL(DonePopping): NESTED_ENTRY RhpCallFilterFunclet, _TEXT, NoHandler ALLOC_CALL_FUNCLET_FRAME 0x60 +#if __riscv_flen >= 64 fsd fs0, 0x00(sp) fsd fs1, 0x08(sp) fsd fs2, 0x10(sp) @@ -697,6 +731,7 @@ LOCAL_LABEL(DonePopping): fsd fs9, 0x48(sp) fsd fs10, 0x50(sp) fsd fs11, 0x58(sp) +#endif ld t3, OFFSETOF__REGDISPLAY__pFP(a2) ld fp, 0(t3) @@ -709,6 +744,7 @@ LOCAL_LABEL(DonePopping): ALTERNATE_ENTRY RhpCallFilterFunclet2 +#if __riscv_flen >= 64 fld fs0, 0x00(sp) fld fs1, 0x08(sp) fld fs2, 0x10(sp) @@ -721,6 +757,7 @@ LOCAL_LABEL(DonePopping): fld fs9, 0x48(sp) fld fs10, 0x50(sp) fld fs11, 0x58(sp) +#endif FREE_CALL_FUNCLET_FRAME 0x60 EPILOG_RETURN @@ -774,7 +811,14 @@ LOCAL_LABEL(DonePopping): addi t3, a5, OFFSETOF__Thread__m_ThreadStateFlags addiw a6, zero, -17 // Mask value (0xFFFFFFEF) +#ifdef __riscv_atomic amoand.w a4, t3, a6 +#else + // No A extension: non-atomic read-modify-write. + lw a4, (t3) + and a4, a4, a6 + sw a4, (t3) +#endif // set preserved regs to the values expected by the funclet RESTORE_PRESERVED_REGISTERS a2 diff --git a/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S b/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S index 8522bc01037412..050770e34b0fc7 100644 --- a/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S +++ b/src/coreclr/nativeaot/Runtime/riscv64/GcProbe.S @@ -39,8 +39,10 @@ sd a2, 0x90(sp) # Save the FP return registers +#if __riscv_flen >= 64 fsd fa0, 0x98(sp) fsd fa1, 0xa0(sp) +#endif # Slot at sp+0xa8 is alignment padding # Perform the rest of the PInvokeTransitionFrame initialization. @@ -64,8 +66,10 @@ ld a2, 0x90(sp) // Restore the FP return registers +#if __riscv_flen >= 64 fld fa0, 0x98(sp) fld fa1, 0xa0(sp) +#endif // Restore callee saved registers EPILOG_RESTORE_REG_PAIR s1, s2, 0x20 diff --git a/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S b/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S index b38480f8113d2b..7e5fce8ced32ba 100644 --- a/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S +++ b/src/coreclr/nativeaot/Runtime/riscv64/UniversalTransition.S @@ -89,6 +89,7 @@ PROLOG_SAVE_REG_PAIR_INDEXED fp, ra, STACK_SIZE # Floating point registers +#if __riscv_flen >= 64 fsd fa0, FLOAT_ARG_OFFSET(sp) fsd fa1, FLOAT_ARG_OFFSET + 0x08(sp) fsd fa2, FLOAT_ARG_OFFSET + 0x10(sp) @@ -97,6 +98,7 @@ fsd fa5, FLOAT_ARG_OFFSET + 0x28(sp) fsd fa6, FLOAT_ARG_OFFSET + 0x30(sp) fsd fa7, FLOAT_ARG_OFFSET + 0x38(sp) +#endif # Space for return block data (0x10 bytes) @@ -113,6 +115,7 @@ #ifdef TRASH_SAVED_ARGUMENT_REGISTERS PREPARE_EXTERNAL_VAR RhpFpTrashValues, a1 +#if __riscv_flen >= 64 fld fa0, 0x00(a1) fld fa1, 0x08(a1) fld fa2, 0x10(a1) @@ -121,6 +124,7 @@ fld fa5, 0x28(a1) fld fa6, 0x30(a1) fld fa7, 0x38(a1) +#endif PREPARE_EXTERNAL_VAR RhpIntegerTrashValues, a1 @@ -143,6 +147,7 @@ ALTERNATE_ENTRY ReturnFrom\FunctionName mv t2, a0 # Restore floating point registers +#if __riscv_flen >= 64 fld fa0, FLOAT_ARG_OFFSET(sp) fld fa1, FLOAT_ARG_OFFSET + 0x08(sp) fld fa2, FLOAT_ARG_OFFSET + 0x10(sp) @@ -151,6 +156,7 @@ ALTERNATE_ENTRY ReturnFrom\FunctionName fld fa5, FLOAT_ARG_OFFSET + 0x28(sp) fld fa6, FLOAT_ARG_OFFSET + 0x30(sp) fld fa7, FLOAT_ARG_OFFSET + 0x38(sp) +#endif # Restore the argument registers ld a0, ARGUMENT_REGISTERS_OFFSET(sp) diff --git a/src/coreclr/pal/inc/unixasmmacrosriscv64.inc b/src/coreclr/pal/inc/unixasmmacrosriscv64.inc index 406074d6f49436..5342ef99465bf9 100644 --- a/src/coreclr/pal/inc/unixasmmacrosriscv64.inc +++ b/src/coreclr/pal/inc/unixasmmacrosriscv64.inc @@ -161,6 +161,9 @@ C_FUNC(\Name): // Reserve 64 bytes of memory before calling SAVE_FLOAT_ARGUMENT_REGISTERS .macro SAVE_FLOAT_ARGUMENT_REGISTERS reg, ofs + // No-op without an FPU; the caller reserves the slots either way, so frame + // offsets are unaffected. Same for the other FP sequences in this file. +#if __riscv_flen >= 64 fsd fa0, (\ofs)(\reg) fsd fa1, (\ofs + 8)(\reg) fsd fa2, (\ofs + 16)(\reg) @@ -169,6 +172,7 @@ C_FUNC(\Name): fsd fa5, (\ofs + 40)(\reg) fsd fa6, (\ofs + 48)(\reg) fsd fa7, (\ofs + 56)(\reg) +#endif .endm // Reserve 64 bytes of memory before calling SAVE_FLOAT_CALLEESAVED_REGISTERS @@ -199,6 +203,7 @@ C_FUNC(\Name): .endm .macro RESTORE_FLOAT_ARGUMENT_REGISTERS reg, ofs +#if __riscv_flen >= 64 fld fa0, (\ofs)(\reg) fld fa1, (\ofs + 8)(\reg) fld fa2, (\ofs + 16)(\reg) @@ -207,6 +212,7 @@ C_FUNC(\Name): fld fa5, (\ofs + 40)(\reg) fld fa6, (\ofs + 48)(\reg) fld fa7, (\ofs + 56)(\reg) +#endif .endm .macro RESTORE_FLOAT_CALLEESAVED_REGISTERS reg, ofs @@ -310,6 +316,7 @@ C_FUNC(\Name): // Save callee-saved floating point registers if requested (fs0-fs11) .if (__PWTB_PushCalleeSavedFloatRegs == 1) +#if __riscv_flen >= 64 fsd fs0, (__PWTB_FloatCalleeSavedRegisters)(sp) fsd fs1, (__PWTB_FloatCalleeSavedRegisters + 8)(sp) fsd fs2, (__PWTB_FloatCalleeSavedRegisters + 16)(sp) @@ -322,6 +329,7 @@ C_FUNC(\Name): fsd fs9, (__PWTB_FloatCalleeSavedRegisters + 72)(sp) fsd fs10, (__PWTB_FloatCalleeSavedRegisters + 80)(sp) fsd fs11, (__PWTB_FloatCalleeSavedRegisters + 88)(sp) +#endif .endif .endm @@ -405,6 +413,7 @@ C_FUNC(\Name): // Save FP callee-saved registers (fs0-fs11 = f8,f9,f18-f27) at offset 0 // RISC-V FP callee-saved: fs0=f8, fs1=f9, fs2-fs11=f18-f27 +#if __riscv_flen >= 64 fsd fs0, 0(sp) // f8 fsd fs1, 8(sp) // f9 fsd fs2, 16(sp) // f18 @@ -417,6 +426,7 @@ C_FUNC(\Name): fsd fs9, 72(sp) // f25 fsd fs10, 80(sp) // f26 fsd fs11, 88(sp) // f27 +#endif // Set target to TransitionBlock pointer addi \target, sp, 160 diff --git a/src/coreclr/pal/src/arch/riscv64/context2.S b/src/coreclr/pal/src/arch/riscv64/context2.S index 5bb06b0ecea702..7914eae9921244 100644 --- a/src/coreclr/pal/src/arch/riscv64/context2.S +++ b/src/coreclr/pal/src/arch/riscv64/context2.S @@ -26,6 +26,7 @@ LEAF_ENTRY RtlRestoreContext, _TEXT //64-bits FPR. addi t0, t4, CONTEXT_FPU_OFFSET +#if __riscv_flen >= 64 fld f0, (CONTEXT_F0)(t0) fld f1, (CONTEXT_F1)(t0) fld f2, (CONTEXT_F2)(t0) @@ -58,9 +59,12 @@ LEAF_ENTRY RtlRestoreContext, _TEXT fld f29, (CONTEXT_F29)(t0) fld f30, (CONTEXT_F30)(t0) fld f31, (CONTEXT_F31)(t0) +#endif lw t1, (CONTEXT_FLOAT_CONTROL_OFFSET)(t0) +#if __riscv_flen != 0 fscsr x0, t1 +#endif LOCAL_LABEL(No_Restore_CONTEXT_FLOATING_POINT): @@ -205,6 +209,7 @@ LOCAL_LABEL(Done_CONTEXT_INTEGER): addi a0, a0, CONTEXT_FPU_OFFSET +#if __riscv_flen >= 64 fsd f0, (CONTEXT_F0)(a0) fsd f1, (CONTEXT_F1)(a0) fsd f2, (CONTEXT_F2)(a0) @@ -237,8 +242,11 @@ LOCAL_LABEL(Done_CONTEXT_INTEGER): fsd f29, (CONTEXT_F29)(a0) fsd f30, (CONTEXT_F30)(a0) fsd f31, (CONTEXT_F31)(a0) +#endif +#if __riscv_flen != 0 frcsr t0 +#endif sd t0, (CONTEXT_FLOAT_CONTROL_OFFSET)(a0) LOCAL_LABEL(Done_CONTEXT_FLOATING_POINT): diff --git a/src/coreclr/runtime/riscv64/WriteBarriers.S b/src/coreclr/runtime/riscv64/WriteBarriers.S index c4a336be1caeb4..4dc36d2822d772 100644 --- a/src/coreclr/runtime/riscv64/WriteBarriers.S +++ b/src/coreclr/runtime/riscv64/WriteBarriers.S @@ -277,6 +277,7 @@ LEAF_END RhpAssignRef, _TEXT LEAF_ENTRY RhpCheckedLockCmpXchg LOCAL_LABEL(CmpXchgRetry): +#ifdef __riscv_atomic // Load the current value at the destination address. lr.d.aqrl t0, (a0) // t0 = *dest (load with acquire-release ordering) // Compare the loaded value with the comparand. @@ -285,6 +286,12 @@ LOCAL_LABEL(CmpXchgRetry): // Attempt to store the exchange value at the destination address. sc.d.rl t1, a1, (a0) // t1 = (store conditional result: 0 if successful, with release ordering) bnez t1, LOCAL_LABEL(CmpXchgRetry) // if store conditional failed, retry +#else + // No A extension: non-atomic compare-and-swap. + ld t0, (a0) + bne t0, a2, LOCAL_LABEL(CmpXchgNoUpdate) + sd a1, (a0) +#endif // See comment at the top of PalInterlockedOperationBarrier method for explanation why this memory // barrier is necessary. @@ -321,7 +328,13 @@ LEAF_END RhpCheckedLockCmpXchg // t1, t6: trashed // LEAF_ENTRY RhpCheckedXchg +#ifdef __riscv_atomic amoswap.d.aqrl t1, a1, (a0) +#else + // No A extension: non-atomic exchange, old value in t1. + ld t1, (a0) + sd a1, (a0) +#endif // See comment at the top of PalInterlockedOperationBarrier method for explanation why this memory // barrier is necessary. diff --git a/src/coreclr/vm/precode.h b/src/coreclr/vm/precode.h index 2b6aef0db9251a..ff5b21fa0a12be 100644 --- a/src/coreclr/vm/precode.h +++ b/src/coreclr/vm/precode.h @@ -394,7 +394,13 @@ struct FixupPrecode static const int FixupCodeOffset = 12; #elif defined(TARGET_RISCV64) static const SIZE_T CodeSize = 32; + // The thunk's second half starts after the tail jump, which is 2 bytes + // shorter with the C extension. Keep in sync with thunktemplates.S. +#ifdef __riscv_compressed static const int FixupCodeOffset = 10; +#else + static const int FixupCodeOffset = 12; +#endif #endif // TARGET_AMD64 BYTE m_code[CodeSize]; diff --git a/src/coreclr/vm/riscv64/asmhelpers.S b/src/coreclr/vm/riscv64/asmhelpers.S index 486fce7790ec56..6e94858f41e277 100644 --- a/src/coreclr/vm/riscv64/asmhelpers.S +++ b/src/coreclr/vm/riscv64/asmhelpers.S @@ -489,8 +489,10 @@ NESTED_ENTRY OnHijackTripThread, _TEXT, NoHandler sd a2, 136(sp) // save any FP/HFA return value(s) +#if __riscv_flen >= 64 fsd f0, 144(sp) fsd f1, 152(sp) +#endif addi a0, sp, 0 call C_FUNC(OnHijackWorker) @@ -504,8 +506,10 @@ NESTED_ENTRY OnHijackTripThread, _TEXT, NoHandler ld a2, 136(sp) // restore any FP/HFA return value(s) +#if __riscv_flen >= 64 fld f0, 144(sp) fld f1, 152(sp) +#endif EPILOG_RESTORE_REG_PAIR s1, s2, 16 EPILOG_RESTORE_REG_PAIR s3, s4, 32 @@ -1206,7 +1210,9 @@ NESTED_ENTRY CallJittedMethodRetDouble, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen >= 64 fsd fa0, 0(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1230,7 +1236,9 @@ NESTED_ENTRY CallJittedMethodRetFloat, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen != 0 fsw fa0, 0(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1279,8 +1287,10 @@ NESTED_ENTRY CallJittedMethodRet2Double, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen >= 64 fsd fa0, 0(a2) fsd fa1, 8(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1304,8 +1314,10 @@ NESTED_ENTRY CallJittedMethodRet2Float, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen != 0 fsw fa0, 0(a2) fsw fa1, 4(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1329,7 +1341,9 @@ NESTED_ENTRY CallJittedMethodRetFloatInt, _TEXT, NoHandler ld a4, 24(fp) sd a2, 0(a4) ld a2, 16(fp) +#if __riscv_flen >= 64 fsd fa0, 0(a2) +#endif sd a0, 8(a2) EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 @@ -1355,7 +1369,9 @@ NESTED_ENTRY CallJittedMethodRetIntFloat, _TEXT, NoHandler sd a2, 0(a4) ld a2, 16(fp) sd a0, 0(a2) +#if __riscv_flen >= 64 fsd fa0, 8(a2) +#endif EPILOG_STACK_RESTORE EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 32 EPILOG_RETURN @@ -1434,7 +1450,9 @@ NESTED_ENTRY InterpreterStubRetDouble, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen >= 64 fld fa0, 0(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRetDouble, _TEXT @@ -1470,8 +1488,10 @@ NESTED_ENTRY InterpreterStubRet2Double, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen >= 64 fld fa0, 0(a0) fld fa1, 8(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRet2Double, _TEXT @@ -1483,7 +1503,9 @@ NESTED_ENTRY InterpreterStubRetFloat, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen != 0 flw fa0, 0(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRetFloat, _TEXT @@ -1495,8 +1517,10 @@ NESTED_ENTRY InterpreterStubRet2Float, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen != 0 flw fa0, 0(a0) flw fa1, 4(a0) +#endif EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN NESTED_END InterpreterStubRet2Float, _TEXT @@ -1508,7 +1532,9 @@ NESTED_ENTRY InterpreterStubRetFloatInt, _TEXT, NoHandler mv a1, s1 // the IR bytecode pointer mv a2, zero call C_FUNC(ExecuteInterpretedMethod) +#if __riscv_flen >= 64 fld fa0, 0(a0) +#endif ld a0, 8(a0) EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN @@ -1522,7 +1548,9 @@ NESTED_ENTRY InterpreterStubRetIntFloat, _TEXT, NoHandler mv a2, zero call C_FUNC(ExecuteInterpretedMethod) ld a1, 0(a0) +#if __riscv_flen >= 64 fld fa0, 8(a0) +#endif mv a0, a1 EPILOG_RESTORE_REG_PAIR_INDEXED fp, ra, 16 EPILOG_RETURN @@ -1979,9 +2007,14 @@ ALTERNATE_ENTRY Store_A7 LEAF_END Store_A1_A2_A3_A4_A5_A6_A7 // Float point load/store routines +// +// The symbols are always defined because the call stub generator references +// them; without an FPU nothing is routed through fa0-fa7, so they are unused. LEAF_ENTRY Load_FA0 +#if __riscv_flen >= 64 fld fa0, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -1989,8 +2022,10 @@ LEAF_ENTRY Load_FA0 LEAF_END Load_FA0 LEAF_ENTRY Load_FA0_FA1 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -1998,11 +2033,15 @@ LEAF_ENTRY Load_FA0_FA1 LEAF_END Load_FA0_FA1 LEAF_ENTRY Load_FA0_FA1_FA2 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2 +#if __riscv_flen >= 64 fld fa2, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2010,12 +2049,16 @@ ALTERNATE_ENTRY Load_FA2 LEAF_END Load_FA0_FA1_FA2 LEAF_ENTRY Load_FA0_FA1_FA2_FA3 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2023,15 +2066,21 @@ ALTERNATE_ENTRY Load_FA2_FA3 LEAF_END Load_FA0_FA1_FA2_FA3 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4 +#if __riscv_flen >= 64 fld fa4, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2039,16 +2088,22 @@ ALTERNATE_ENTRY Load_FA4 LEAF_END Load_FA0_FA1_FA2_FA3_FA4 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4_FA5 +#if __riscv_flen >= 64 fld fa4, 0(t3) fld fa5, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2056,19 +2111,27 @@ ALTERNATE_ENTRY Load_FA4_FA5 LEAF_END Load_FA0_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa4, 0(t3) fld fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA6 +#if __riscv_flen >= 64 fld fa6, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2076,20 +2139,28 @@ ALTERNATE_ENTRY Load_FA6 LEAF_END Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa0, 0(t3) fld fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa2, 0(t3) fld fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa4, 0(t3) fld fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Load_FA6_FA7 +#if __riscv_flen >= 64 fld fa6, 0(t3) fld fa7, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2098,7 +2169,9 @@ LEAF_END Load_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 // Additional Load_FA* routines starting from FA1, FA3, FA5, FA7 LEAF_ENTRY Load_FA1 +#if __riscv_flen >= 64 fld fa1, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2106,8 +2179,10 @@ LEAF_ENTRY Load_FA1 LEAF_END Load_FA1 LEAF_ENTRY Load_FA1_FA2 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2115,9 +2190,11 @@ LEAF_ENTRY Load_FA1_FA2 LEAF_END Load_FA1_FA2 LEAF_ENTRY Load_FA1_FA2_FA3 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2125,10 +2202,12 @@ LEAF_ENTRY Load_FA1_FA2_FA3 LEAF_END Load_FA1_FA2_FA3 LEAF_ENTRY Load_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) fld fa4, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2136,11 +2215,13 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4 LEAF_END Load_FA1_FA2_FA3_FA4 LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) fld fa4, 24(t3) fld fa5, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2148,12 +2229,14 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5 LEAF_END Load_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) fld fa4, 24(t3) fld fa5, 32(t3) fld fa6, 40(t3) +#endif addi t3, t3, 48 ld t4, 0(t2) addi t2, t2, 8 @@ -2161,6 +2244,7 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_END Load_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa1, 0(t3) fld fa2, 8(t3) fld fa3, 16(t3) @@ -2168,6 +2252,7 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 fld fa5, 32(t3) fld fa6, 40(t3) fld fa7, 48(t3) +#endif addi t3, t3, 56 ld t4, 0(t2) addi t2, t2, 8 @@ -2175,7 +2260,9 @@ LEAF_ENTRY Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_END Load_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Load_FA3 +#if __riscv_flen >= 64 fld fa3, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2183,8 +2270,10 @@ LEAF_ENTRY Load_FA3 LEAF_END Load_FA3 LEAF_ENTRY Load_FA3_FA4 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2192,9 +2281,11 @@ LEAF_ENTRY Load_FA3_FA4 LEAF_END Load_FA3_FA4 LEAF_ENTRY Load_FA3_FA4_FA5 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) fld fa5, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2202,10 +2293,12 @@ LEAF_ENTRY Load_FA3_FA4_FA5 LEAF_END Load_FA3_FA4_FA5 LEAF_ENTRY Load_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) fld fa5, 16(t3) fld fa6, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2213,11 +2306,13 @@ LEAF_ENTRY Load_FA3_FA4_FA5_FA6 LEAF_END Load_FA3_FA4_FA5_FA6 LEAF_ENTRY Load_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa3, 0(t3) fld fa4, 8(t3) fld fa5, 16(t3) fld fa6, 24(t3) fld fa7, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2225,7 +2320,9 @@ LEAF_ENTRY Load_FA3_FA4_FA5_FA6_FA7 LEAF_END Load_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Load_FA5 +#if __riscv_flen >= 64 fld fa5, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2233,8 +2330,10 @@ LEAF_ENTRY Load_FA5 LEAF_END Load_FA5 LEAF_ENTRY Load_FA5_FA6 +#if __riscv_flen >= 64 fld fa5, 0(t3) fld fa6, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2242,9 +2341,11 @@ LEAF_ENTRY Load_FA5_FA6 LEAF_END Load_FA5_FA6 LEAF_ENTRY Load_FA5_FA6_FA7 +#if __riscv_flen >= 64 fld fa5, 0(t3) fld fa6, 8(t3) fld fa7, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2252,7 +2353,9 @@ LEAF_ENTRY Load_FA5_FA6_FA7 LEAF_END Load_FA5_FA6_FA7 LEAF_ENTRY Load_FA7 +#if __riscv_flen >= 64 fld fa7, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2260,7 +2363,9 @@ LEAF_ENTRY Load_FA7 LEAF_END Load_FA7 LEAF_ENTRY Store_FA0 +#if __riscv_flen >= 64 fsd fa0, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2268,8 +2373,10 @@ LEAF_ENTRY Store_FA0 LEAF_END Store_FA0 LEAF_ENTRY Store_FA0_FA1 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2277,11 +2384,15 @@ LEAF_ENTRY Store_FA0_FA1 LEAF_END Store_FA0_FA1 LEAF_ENTRY Store_FA0_FA1_FA2 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2 +#if __riscv_flen >= 64 fsd fa2, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2289,12 +2400,16 @@ ALTERNATE_ENTRY Store_FA2 LEAF_END Store_FA0_FA1_FA2 LEAF_ENTRY Store_FA0_FA1_FA2_FA3 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2302,15 +2417,21 @@ ALTERNATE_ENTRY Store_FA2_FA3 LEAF_END Store_FA0_FA1_FA2_FA3 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4 +#if __riscv_flen >= 64 fsd fa4, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2318,16 +2439,22 @@ ALTERNATE_ENTRY Store_FA4 LEAF_END Store_FA0_FA1_FA2_FA3_FA4 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4_FA5 +#if __riscv_flen >= 64 fsd fa4, 0(t3) fsd fa5, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2335,19 +2462,27 @@ ALTERNATE_ENTRY Store_FA4_FA5 LEAF_END Store_FA0_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa4, 0(t3) fsd fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA6 +#if __riscv_flen >= 64 fsd fa6, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2355,20 +2490,28 @@ ALTERNATE_ENTRY Store_FA6 LEAF_END Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa0, 0(t3) fsd fa1, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa2, 0(t3) fsd fa3, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa4, 0(t3) fsd fa5, 8(t3) +#endif addi t3, t3, 16 ALTERNATE_ENTRY Store_FA6_FA7 +#if __riscv_flen >= 64 fsd fa6, 0(t3) fsd fa7, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2377,7 +2520,9 @@ LEAF_END Store_FA0_FA1_FA2_FA3_FA4_FA5_FA6_FA7 // Additional Store_FA* routines starting from FA1, FA3, FA5, FA7 LEAF_ENTRY Store_FA1 +#if __riscv_flen >= 64 fsd fa1, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2385,8 +2530,10 @@ LEAF_ENTRY Store_FA1 LEAF_END Store_FA1 LEAF_ENTRY Store_FA1_FA2 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2394,9 +2541,11 @@ LEAF_ENTRY Store_FA1_FA2 LEAF_END Store_FA1_FA2 LEAF_ENTRY Store_FA1_FA2_FA3 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2404,10 +2553,12 @@ LEAF_ENTRY Store_FA1_FA2_FA3 LEAF_END Store_FA1_FA2_FA3 LEAF_ENTRY Store_FA1_FA2_FA3_FA4 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) fsd fa4, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2415,11 +2566,13 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4 LEAF_END Store_FA1_FA2_FA3_FA4 LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) fsd fa4, 24(t3) fsd fa5, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2427,12 +2580,14 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5 LEAF_END Store_FA1_FA2_FA3_FA4_FA5 LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) fsd fa4, 24(t3) fsd fa5, 32(t3) fsd fa6, 40(t3) +#endif addi t3, t3, 48 ld t4, 0(t2) addi t2, t2, 8 @@ -2440,6 +2595,7 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_END Store_FA1_FA2_FA3_FA4_FA5_FA6 LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa1, 0(t3) fsd fa2, 8(t3) fsd fa3, 16(t3) @@ -2447,6 +2603,7 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 fsd fa5, 32(t3) fsd fa6, 40(t3) fsd fa7, 48(t3) +#endif addi t3, t3, 56 ld t4, 0(t2) addi t2, t2, 8 @@ -2454,7 +2611,9 @@ LEAF_ENTRY Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_END Store_FA1_FA2_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Store_FA3 +#if __riscv_flen >= 64 fsd fa3, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2462,8 +2621,10 @@ LEAF_ENTRY Store_FA3 LEAF_END Store_FA3 LEAF_ENTRY Store_FA3_FA4 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2471,9 +2632,11 @@ LEAF_ENTRY Store_FA3_FA4 LEAF_END Store_FA3_FA4 LEAF_ENTRY Store_FA3_FA4_FA5 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) fsd fa5, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2481,10 +2644,12 @@ LEAF_ENTRY Store_FA3_FA4_FA5 LEAF_END Store_FA3_FA4_FA5 LEAF_ENTRY Store_FA3_FA4_FA5_FA6 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) fsd fa5, 16(t3) fsd fa6, 24(t3) +#endif addi t3, t3, 32 ld t4, 0(t2) addi t2, t2, 8 @@ -2492,11 +2657,13 @@ LEAF_ENTRY Store_FA3_FA4_FA5_FA6 LEAF_END Store_FA3_FA4_FA5_FA6 LEAF_ENTRY Store_FA3_FA4_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa3, 0(t3) fsd fa4, 8(t3) fsd fa5, 16(t3) fsd fa6, 24(t3) fsd fa7, 32(t3) +#endif addi t3, t3, 40 ld t4, 0(t2) addi t2, t2, 8 @@ -2504,7 +2671,9 @@ LEAF_ENTRY Store_FA3_FA4_FA5_FA6_FA7 LEAF_END Store_FA3_FA4_FA5_FA6_FA7 LEAF_ENTRY Store_FA5 +#if __riscv_flen >= 64 fsd fa5, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 @@ -2512,8 +2681,10 @@ LEAF_ENTRY Store_FA5 LEAF_END Store_FA5 LEAF_ENTRY Store_FA5_FA6 +#if __riscv_flen >= 64 fsd fa5, 0(t3) fsd fa6, 8(t3) +#endif addi t3, t3, 16 ld t4, 0(t2) addi t2, t2, 8 @@ -2521,9 +2692,11 @@ LEAF_ENTRY Store_FA5_FA6 LEAF_END Store_FA5_FA6 LEAF_ENTRY Store_FA5_FA6_FA7 +#if __riscv_flen >= 64 fsd fa5, 0(t3) fsd fa6, 8(t3) fsd fa7, 16(t3) +#endif addi t3, t3, 24 ld t4, 0(t2) addi t2, t2, 8 @@ -2531,7 +2704,9 @@ LEAF_ENTRY Store_FA5_FA6_FA7 LEAF_END Store_FA5_FA6_FA7 LEAF_ENTRY Store_FA7 +#if __riscv_flen >= 64 fsd fa7, 0(t3) +#endif addi t3, t3, 8 ld t4, 0(t2) addi t2, t2, 8 diff --git a/src/coreclr/vm/riscv64/calldescrworkerriscv64.S b/src/coreclr/vm/riscv64/calldescrworkerriscv64.S index 54725758b41b27..73e901a15e11ca 100644 --- a/src/coreclr/vm/riscv64/calldescrworkerriscv64.S +++ b/src/coreclr/vm/riscv64/calldescrworkerriscv64.S @@ -44,6 +44,7 @@ LOCAL_LABEL(donestack): ld t4, CallDescrData__pFloatArgumentRegisters(s1) beq t4, zero, LOCAL_LABEL(NoFloatingPoint) +#if __riscv_flen >= 64 fld fa0, 0(t4) fld fa1, 8(t4) fld fa2, 16(t4) @@ -52,6 +53,7 @@ LOCAL_LABEL(donestack): fld fa5, 40(t4) fld fa6, 48(t4) fld fa7, 56(t4) +#endif LOCAL_LABEL(NoFloatingPoint): // Copy [pArgumentRegisters, ..., pArgumentRegisters + 56] @@ -80,7 +82,9 @@ LOCAL_LABEL(CallDescrWorkerInternalReturnAddress): // Just save the returned registers (fa0, fa1/a0) and let CopyReturnedFpStructFromRegisters worry about placing // the fields as they were originally laid out in memory. +#if __riscv_flen >= 64 fsd fa0, CallDescrData__returnValue(s1) // fa0 is always occupied; we have at least one floating field +#endif andi a3, a3, FpStruct__BothFloat bne a3, zero, LOCAL_LABEL(SecondFieldFloatReturn) @@ -91,7 +95,9 @@ LOCAL_LABEL(CallDescrWorkerInternalReturnAddress): j LOCAL_LABEL(ReturnDone) LOCAL_LABEL(SecondFieldFloatReturn): +#if __riscv_flen >= 64 fsd fa1, (CallDescrData__returnValue + 8)(s1) +#endif j LOCAL_LABEL(ReturnDone) LOCAL_LABEL(IntReturn): diff --git a/src/coreclr/vm/riscv64/thunktemplates.S b/src/coreclr/vm/riscv64/thunktemplates.S index ce4e1d4b912025..34fae32559e42d 100644 --- a/src/coreclr/vm/riscv64/thunktemplates.S +++ b/src/coreclr/vm/riscv64/thunktemplates.S @@ -14,6 +14,9 @@ LEAF_END_MARKED StubPrecodeCode LEAF_ENTRY FixupPrecodeCode auipc t2, 0x4 ld t2, (FixupPrecodeData__Target)(t2) + // Without the C extension the tail jump is 4 bytes instead of 2, shifting + // the rest of the thunk by 2. Keep in sync with FixupPrecode::FixupCodeOffset. +#ifdef __riscv_compressed c.jr t2 fence r,rw @@ -21,6 +24,15 @@ LEAF_ENTRY FixupPrecodeCode ld t1, (FixupPrecodeData__PrecodeFixupThunk - 0xe)(t2) ld t2, (FixupPrecodeData__MethodDesc - 0xe)(t2) jr t1 +#else + jr t2 + + fence r,rw + auipc t2, 0x4 + ld t1, (FixupPrecodeData__PrecodeFixupThunk - 0x10)(t2) + ld t2, (FixupPrecodeData__MethodDesc - 0x10)(t2) + jr t1 +#endif LEAF_END_MARKED FixupPrecodeCode #ifdef FEATURE_TIERED_COMPILATION From f63476a7621a12c62172f177daefcdb10930638e Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 9 Sep 2026 11:26:55 +0100 Subject: [PATCH 07/16] [RISC-V] Tie the hwprobe assertion to the build's own -march cpufeatures.c asserts the full rv64gc baseline at startup, so a runtime built for a narrower ISA fails the assert on hardware that matches what it was built for. Assert what the build actually requires instead. Signed-off-by: Maxim Menshikov --- src/native/minipal/cpufeatures.c | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/src/native/minipal/cpufeatures.c b/src/native/minipal/cpufeatures.c index ed7d7612779842..f400388f0cc9fb 100644 --- a/src/native/minipal/cpufeatures.c +++ b/src/native/minipal/cpufeatures.c @@ -775,9 +775,16 @@ int minipal_getcpufeatures(void) if (syscall(__NR_riscv_hwprobe, pairs, 1, 0, NULL, 0) == 0) { - // Our baseline support is for RV64GC (see #73437) + // The hardware must implement at least the ISA the runtime was built + // for. RV64GC is the supported baseline (see #73437); a build for a + // reduced -march only assumes the extensions it was compiled with. + // (hwprobe has no F-only bit, so an F-without-D build asserts nothing.) +#if defined(__riscv_flen) && __riscv_flen >= 64 assert(pairs[0].value & RISCV_HWPROBE_IMA_FD); +#endif +#ifdef __riscv_compressed assert(pairs[0].value & RISCV_HWPROBE_IMA_C); +#endif if (pairs[0].value & RISCV_HWPROBE_EXT_ZBA) { From 639985408787cd7f58e4c300af43d547c118c82e Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 9 Sep 2026 11:27:22 +0100 Subject: [PATCH 08/16] [RISC-V] Add the lp64 soft-float ABI to the JIT Under CORJIT_FLAG_SOFTFP_ABI - the flag armel already uses - riscv64 passes and returns floating-point scalars and floating-point struct fields by the integer calling convention, which is what the lp64 psABI prescribes for a target without F/D. Only the ABI classifier changes here; code generation still uses the FP registers, so this commit alone is not a usable target. It is separated because it is the part that is purely a calling convention. Signed-off-by: Maxim Menshikov --- src/coreclr/inc/corjitflags.h | 4 +++- src/coreclr/jit/compiler.cpp | 7 ++++++- src/coreclr/jit/compiler.h | 4 +++- src/coreclr/jit/gentree.cpp | 16 ++++++++++------ src/coreclr/jit/jitee.h | 8 ++++++-- src/coreclr/jit/targetriscv64.cpp | 4 ++-- 6 files changed, 30 insertions(+), 13 deletions(-) diff --git a/src/coreclr/inc/corjitflags.h b/src/coreclr/inc/corjitflags.h index d898fcda5a2e59..34d9338997366a 100644 --- a/src/coreclr/inc/corjitflags.h +++ b/src/coreclr/inc/corjitflags.h @@ -61,7 +61,9 @@ class CORJIT_FLAGS #if defined(TARGET_ARM) CORJIT_FLAG_RELATIVE_CODE_RELOCS = 29, // JIT should generate PC-relative address computations instead of EE relocation records - CORJIT_FLAG_SOFTFP_ABI = 30, // Enable armel calling convention +#endif +#if defined(TARGET_ARM) || defined(TARGET_RISCV64) + CORJIT_FLAG_SOFTFP_ABI = 30, // Enable the soft-float calling convention (armel; lp64 on RISC-V) #endif CORJIT_FLAG_USE_DISPATCH_HELPERS = 31, // The JIT should use helpers for interface dispatch instead of virtual stub dispatch CORJIT_FLAG_VERIFY_GC_MODE_TRANSITIONS = 32, // The JIT should emit the diagnostic helpers that verify GC mode transitions are legal diff --git a/src/coreclr/jit/compiler.cpp b/src/coreclr/jit/compiler.cpp index fdca6605168ff4..c907297271cb46 100644 --- a/src/coreclr/jit/compiler.cpp +++ b/src/coreclr/jit/compiler.cpp @@ -867,7 +867,7 @@ var_types Compiler::getReturnTypeForStruct(CORINFO_CLASS_HANDLE clsHnd, useType = TYP_UNKNOWN; } #elif defined(TARGET_RISCV64) || defined(TARGET_LOONGARCH64) - if (structSize <= (TARGET_POINTER_SIZE * 2)) + if ((structSize <= (TARGET_POINTER_SIZE * 2)) && !opts.compUseSoftFP) { const CORINFO_FPSTRUCT_LOWERING* lowering = GetFpStructLowering(clsHnd); if (!lowering->byIntegerCallConv) @@ -2906,6 +2906,11 @@ void Compiler::compInitOptions(JitFlags* jitFlags) } GlobalJitOptions::compFeatureHfa = !opts.compUseSoftFP; +#elif defined(TARGET_RISCV64) + // lp64 soft-float ABI: FP scalars and FP struct fields are passed by the + // integer calling convention and fa* registers are never used. Set by the + // VM / AOT driver for targets without the F extension. + opts.compUseSoftFP = jitFlags->IsSet(JitFlags::JIT_FLAG_SOFTFP_ABI); #elif defined(ARM_SOFTFP) && defined(TARGET_ARM) // Armel is unconditionally enabled in the JIT. Verify that the VM side agrees. assert(jitFlags->IsSet(JitFlags::JIT_FLAG_SOFTFP_ABI)); diff --git a/src/coreclr/jit/compiler.h b/src/coreclr/jit/compiler.h index ec2b6d751e0584..a355f05a402625 100644 --- a/src/coreclr/jit/compiler.h +++ b/src/coreclr/jit/compiler.h @@ -11653,7 +11653,9 @@ class Compiler int compJitSaveFpLrWithCalleeSavedRegisters; #endif // defined(TARGET_ARM64) -#ifdef CONFIGURABLE_ARM_ABI +#if defined(CONFIGURABLE_ARM_ABI) || defined(TARGET_RISCV64) + // On RISCV64 the lp64 soft-float ABI is selected per compilation by + // JIT_FLAG_SOFTFP_ABI (no-F targets); see compInitOptions. bool compUseSoftFP = false; #else #ifdef ARM_SOFTFP diff --git a/src/coreclr/jit/gentree.cpp b/src/coreclr/jit/gentree.cpp index 9b72cec1087afc..dbb95c462526e0 100644 --- a/src/coreclr/jit/gentree.cpp +++ b/src/coreclr/jit/gentree.cpp @@ -34297,11 +34297,14 @@ void ReturnTypeDesc::InitializeStructReturnType(Compiler* comp, m_regType[0] = returnType; #if defined(TARGET_RISCV64) || defined(TARGET_LOONGARCH64) - const CORINFO_FPSTRUCT_LOWERING* lowering = comp->GetFpStructLowering(retClsHnd); - if (!lowering->byIntegerCallConv) + if (!comp->opts.compUseSoftFP) { - assert(lowering->numLoweredElements == 1); - m_fieldOffset[0] = lowering->offsets[0]; + const CORINFO_FPSTRUCT_LOWERING* lowering = comp->GetFpStructLowering(retClsHnd); + if (!lowering->byIntegerCallConv) + { + assert(lowering->numLoweredElements == 1); + m_fieldOffset[0] = lowering->offsets[0]; + } } #endif // defined(TARGET_RISCV64) || defined(TARGET_LOONGARCH64) break; @@ -34373,8 +34376,9 @@ void ReturnTypeDesc::InitializeStructReturnType(Compiler* comp, assert(structSize <= (2 * TARGET_POINTER_SIZE)); BYTE gcPtrs[2] = {TYPE_GC_NONE, TYPE_GC_NONE}; comp->info.compCompHnd->getClassGClayout(retClsHnd, &gcPtrs[0]); - const CORINFO_FPSTRUCT_LOWERING* lowering = comp->GetFpStructLowering(retClsHnd); - if (!lowering->byIntegerCallConv) + const CORINFO_FPSTRUCT_LOWERING* lowering = + comp->opts.compUseSoftFP ? nullptr : comp->GetFpStructLowering(retClsHnd); + if ((lowering != nullptr) && !lowering->byIntegerCallConv) { comp->compFloatingPointUsed = true; assert(lowering->numLoweredElements == MAX_RET_REG_COUNT); diff --git a/src/coreclr/jit/jitee.h b/src/coreclr/jit/jitee.h index 488d012f77babe..231020da41cce7 100644 --- a/src/coreclr/jit/jitee.h +++ b/src/coreclr/jit/jitee.h @@ -40,7 +40,9 @@ class JitFlags #if defined(TARGET_ARM) JIT_FLAG_RELATIVE_CODE_RELOCS = 29, // JIT should generate PC-relative address computations instead of EE relocation records - JIT_FLAG_SOFTFP_ABI = 30, // Enable armel calling convention +#endif +#if defined(TARGET_ARM) || defined(TARGET_RISCV64) + JIT_FLAG_SOFTFP_ABI = 30, // Enable the soft-float calling convention (armel; lp64 on RISC-V) #endif JIT_FLAG_USE_DISPATCH_HELPERS = 31, // The JIT should use helpers for interface dispatch instead of virtual stub dispatch @@ -141,8 +143,10 @@ class JitFlags #if defined(TARGET_ARM) FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_RELATIVE_CODE_RELOCS, JIT_FLAG_RELATIVE_CODE_RELOCS); - FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_SOFTFP_ABI, JIT_FLAG_SOFTFP_ABI); #endif // TARGET_ARM +#if defined(TARGET_ARM) || defined(TARGET_RISCV64) + FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_SOFTFP_ABI, JIT_FLAG_SOFTFP_ABI); +#endif // TARGET_ARM || TARGET_RISCV64 FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_ASYNC, JIT_FLAG_ASYNC); FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_USE_DISPATCH_HELPERS, JIT_FLAG_USE_DISPATCH_HELPERS); FLAGS_EQUAL(CORJIT_FLAGS::CORJIT_FLAG_VERIFY_GC_MODE_TRANSITIONS, JIT_FLAG_VERIFY_GC_MODE_TRANSITIONS); diff --git a/src/coreclr/jit/targetriscv64.cpp b/src/coreclr/jit/targetriscv64.cpp index 5ba2a3100b90e8..b60807496244f3 100644 --- a/src/coreclr/jit/targetriscv64.cpp +++ b/src/coreclr/jit/targetriscv64.cpp @@ -78,7 +78,7 @@ ABIPassingInformation RiscV64Classifier::Classify(Compiler* comp, passedByRef = true; passedSize = TARGET_POINTER_SIZE; } - else if (!structLayout->IsBlockLayout()) + else if (!structLayout->IsBlockLayout() && !comp->opts.compUseSoftFP) { lowering = comp->GetFpStructLowering(structLayout->GetClassHandle()); if (!lowering->byIntegerCallConv) @@ -100,7 +100,7 @@ ABIPassingInformation RiscV64Classifier::Classify(Compiler* comp, { passedSize = genTypeSize(type); assert(passedSize <= TARGET_POINTER_SIZE); - floatFields = varTypeIsFloating(type) ? 1 : 0; + floatFields = (varTypeIsFloating(type) && !comp->opts.compUseSoftFP) ? 1 : 0; } assert((floatFields > 0) || (intFields == 0)); From 3edf61219aacadf6a4aa744068070bd3bee0ab2c Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 16 Sep 2026 15:26:14 +0100 Subject: [PATCH 09/16] [RISC-V] Add the soft-float helpers and the riscv64-lp64 target Fourteen helpers - {FLT,DBL}{ADD,SUB,MUL,DIV}, {FLT,DBL}CMP_{LE,GE}, FLT2DBL and DBL2FLT - bound by ilc to the toolchain's compiler-rt builtins (__adddf3, __ledf2, ...), the same way CORINFO_HELP_DBLREM is already bound to fmod. No floating-point arithmetic is implemented in the runtime or the libraries. The target is selected explicitly as TargetAbi.NativeAotRiscV64SoftFloat, spelled riscv64-lp64 on the ilc command line, following the armel precedent. It drives the JIT flag, the RISC-V ELF float-ABI field in e_flags (which is what lets the linker reject a mix of lp64 and lp64d objects), the instruction-set defaults and a two-way validation: lp64d requires F and D, lp64 must not have them. crossgen2 rejects the target - the helpers have no ReadyToRun encoding. Value numbering models the helpers as the operations they implement, as CORINFO_HELP_LMUL is modelled on 32-bit targets; only the two three-way compares need VNFuncs of their own. Note: the RiscV64ObjectWriterTests addition was dropped - ILCompiler.Compiler.Tests was deleted upstream in #133474 and needs a new home. Signed-off-by: Maxim Menshikov --- src/coreclr/inc/corinfo.h | 22 +++++++++ src/coreclr/inc/jithelpers.h | 18 ++++++++ src/coreclr/jit/utils.cpp | 15 ++++++ src/coreclr/jit/valuenum.cpp | 46 +++++++++++++++++++ src/coreclr/jit/valuenumfuncs.h | 6 +++ .../tools/Common/CommandLineHelpers.cs | 5 +- .../Compiler/ObjectWriter/ElfObjectWriter.cs | 23 +++++++++- .../ObjectWriter/ObjectWritingOptions.cs | 4 ++ .../tools/Common/InstructionSetHelpers.cs | 10 ++-- .../Internal/Runtime/ReadyToRunConstants.cs | 16 +++++++ .../Common/JitInterface/CorInfoHelpFunc.cs | 16 +++++++ .../tools/Common/JitInterface/CorInfoImpl.cs | 7 +++ .../Common/TypeSystem/Common/TargetDetails.cs | 5 ++ .../ILCompiler.Compiler/Compiler/JitHelper.cs | 44 ++++++++++++++++++ .../ILCompiler.Diagnostics/PerfMapWriter.cs | 1 + .../TestCases/R2RTestSuites.cs | 23 ++++++++++ .../TestCasesRunner/R2RTestRunner.cs | 23 +++++++++- .../Compiler/RyuJitCompilation.cs | 24 ++++++++++ .../JitInterface/CorInfoImpl.RyuJit.cs | 42 +++++++++++++++++ .../aot/ILCompiler/ILCompilerRootCommand.cs | 4 +- src/coreclr/tools/aot/ILCompiler/Program.cs | 3 +- src/coreclr/tools/aot/crossgen2/Program.cs | 8 +++- 22 files changed, 355 insertions(+), 10 deletions(-) diff --git a/src/coreclr/inc/corinfo.h b/src/coreclr/inc/corinfo.h index cd16f8abbbd3d9..c11c5dddcc69bc 100644 --- a/src/coreclr/inc/corinfo.h +++ b/src/coreclr/inc/corinfo.h @@ -342,6 +342,28 @@ enum CorInfoHelpFunc CORINFO_HELP_FLTREM, CORINFO_HELP_DBLREM, + /* Soft-float helpers, for targets without a floating-point unit (RISC-V + without the F/D extensions). The arithmetic helpers have the same + semantics as the corresponding IL instructions. The compare helpers + return a three-way result (< 0, 0, > 0 for less, equal, greater) and + differ only for unordered operands: CMP_LE returns 1 and CMP_GE returns + -1 (the libgcc __le*f2/__ge*f2 conventions), so that every ordered and + unordered IL comparison maps onto a single call. */ + CORINFO_HELP_FLTADD, + CORINFO_HELP_FLTSUB, + CORINFO_HELP_FLTMUL, + CORINFO_HELP_FLTDIV, + CORINFO_HELP_DBLADD, + CORINFO_HELP_DBLSUB, + CORINFO_HELP_DBLMUL, + CORINFO_HELP_DBLDIV, + CORINFO_HELP_FLTCMP_LE, + CORINFO_HELP_FLTCMP_GE, + CORINFO_HELP_DBLCMP_LE, + CORINFO_HELP_DBLCMP_GE, + CORINFO_HELP_FLT2DBL, + CORINFO_HELP_DBL2FLT, + /* Allocating a new object. Always use ICorClassInfo::getNewHelper() to decide which is the right helper to use to allocate an object of a given type. */ diff --git a/src/coreclr/inc/jithelpers.h b/src/coreclr/inc/jithelpers.h index 99a84286242c8f..bad88c5fab87f5 100644 --- a/src/coreclr/inc/jithelpers.h +++ b/src/coreclr/inc/jithelpers.h @@ -97,6 +97,24 @@ JITHELPER(CORINFO_HELP_FLTREM, JIT_FltRem, METHOD__NIL) JITHELPER(CORINFO_HELP_DBLREM, JIT_DblRem, METHOD__NIL) + // Soft-float helpers. The JIT only requests them under CORJIT_FLAG_SOFTFP_ABI on + // targets without an FPU (NativeAOT for RISC-V without F/D, where the AOT + // compiler binds them to the compiler-rt builtins); the VM never sets that flag. + JITHELPER(CORINFO_HELP_FLTADD, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTSUB, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTMUL, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTDIV, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLADD, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLSUB, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLMUL, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLDIV, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTCMP_LE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLTCMP_GE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLCMP_LE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBLCMP_GE, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_FLT2DBL, NULL, METHOD__NIL) + JITHELPER(CORINFO_HELP_DBL2FLT, NULL, METHOD__NIL) + // Allocating a new object JITHELPER(CORINFO_HELP_NEWFAST, RhpNew, METHOD__NIL) JITHELPER(CORINFO_HELP_NEWFAST_MAYBEFROZEN, RhpNewMaybeFrozen, METHOD__NIL) diff --git a/src/coreclr/jit/utils.cpp b/src/coreclr/jit/utils.cpp index 17263b34b48ec5..5e835a26bc190f 100644 --- a/src/coreclr/jit/utils.cpp +++ b/src/coreclr/jit/utils.cpp @@ -1458,6 +1458,21 @@ void HelperCallProperties::init() case CORINFO_HELP_LLSH: case CORINFO_HELP_LRSH: case CORINFO_HELP_LRSZ: + // Soft-float helpers: leaf compiler-rt routines, no GC interaction. + case CORINFO_HELP_FLTADD: + case CORINFO_HELP_FLTSUB: + case CORINFO_HELP_FLTMUL: + case CORINFO_HELP_FLTDIV: + case CORINFO_HELP_DBLADD: + case CORINFO_HELP_DBLSUB: + case CORINFO_HELP_DBLMUL: + case CORINFO_HELP_DBLDIV: + case CORINFO_HELP_FLTCMP_LE: + case CORINFO_HELP_FLTCMP_GE: + case CORINFO_HELP_DBLCMP_LE: + case CORINFO_HELP_DBLCMP_GE: + case CORINFO_HELP_FLT2DBL: + case CORINFO_HELP_DBL2FLT: isNoGC = true; FALLTHROUGH; case CORINFO_HELP_LMUL: diff --git a/src/coreclr/jit/valuenum.cpp b/src/coreclr/jit/valuenum.cpp index 95ab658054d937..88fcac3c02e44d 100644 --- a/src/coreclr/jit/valuenum.cpp +++ b/src/coreclr/jit/valuenum.cpp @@ -15206,6 +15206,14 @@ void Compiler::fgValueNumberCastHelper(GenTreeCall* call) castFromType = TYP_DOUBLE; hasOverflowCheck = true; break; + case CORINFO_HELP_FLT2DBL: + castToType = TYP_DOUBLE; + castFromType = TYP_FLOAT; + break; + case CORINFO_HELP_DBL2FLT: + castToType = TYP_FLOAT; + castFromType = TYP_DOUBLE; + break; default: unreached(); @@ -15273,6 +15281,42 @@ VNFunc Compiler::fgValueNumberJitHelperMethodVNFunc(CorInfoHelpFunc helpFunc) case CORINFO_HELP_DBLREM: vnf = VNF_MOD; break; + case CORINFO_HELP_FLTADD: + vnf = VNFunc(GT_ADD); + break; + case CORINFO_HELP_DBLADD: + vnf = VNFunc(GT_ADD); + break; + case CORINFO_HELP_FLTSUB: + vnf = VNFunc(GT_SUB); + break; + case CORINFO_HELP_DBLSUB: + vnf = VNFunc(GT_SUB); + break; + case CORINFO_HELP_FLTMUL: + vnf = VNFunc(GT_MUL); + break; + case CORINFO_HELP_DBLMUL: + vnf = VNFunc(GT_MUL); + break; + case CORINFO_HELP_FLTDIV: + vnf = VNFunc(GT_DIV); + break; + case CORINFO_HELP_DBLDIV: + vnf = VNFunc(GT_DIV); + break; + case CORINFO_HELP_FLTCMP_LE: + vnf = VNF_SoftFPCmpLE; + break; + case CORINFO_HELP_DBLCMP_LE: + vnf = VNF_SoftFPCmpLE; + break; + case CORINFO_HELP_FLTCMP_GE: + vnf = VNF_SoftFPCmpGE; + break; + case CORINFO_HELP_DBLCMP_GE: + vnf = VNF_SoftFPCmpGE; + break; // These allocation operations probably require some augmentation -- perhaps allocSiteId, // something about array length... @@ -15515,6 +15559,8 @@ bool Compiler::fgValueNumberHelperCall(GenTreeCall* call) case CORINFO_HELP_DBL2LNG: case CORINFO_HELP_DBL2LNG_OVF: case CORINFO_HELP_DBL2UINT_OVF: + case CORINFO_HELP_FLT2DBL: + case CORINFO_HELP_DBL2FLT: case CORINFO_HELP_DBL2ULNG: case CORINFO_HELP_DBL2ULNG_OVF: fgValueNumberCastHelper(call); diff --git a/src/coreclr/jit/valuenumfuncs.h b/src/coreclr/jit/valuenumfuncs.h index f063957255e964..647103a06ea2dc 100644 --- a/src/coreclr/jit/valuenumfuncs.h +++ b/src/coreclr/jit/valuenumfuncs.h @@ -208,6 +208,12 @@ ValueNumFuncDef(HWI_INTRINSIC_END, -1, false, false) #define VNF_HWI_LAST (VNF_HWI_INTRINSIC_END - 1) #endif // FEATURE_HW_INTRINSICS + // Three-way floating-point comparisons of the soft-float compare helpers + // (CORINFO_HELP_FLTCMP_LE & co). Not commutative: swapping the operands + // negates the result, and the two forms differ for unordered operands. + ValueNumFuncDef(SoftFPCmpLE, 2, false, false) + ValueNumFuncDef(SoftFPCmpGE, 2, false, false) + #if defined(TARGET_RISCV64) // Signed/Unsigned integer min/max intrinsics ValueNumFuncDef(MinInt, 2, true, false) diff --git a/src/coreclr/tools/Common/CommandLineHelpers.cs b/src/coreclr/tools/Common/CommandLineHelpers.cs index 58ff170a7e5252..0d917dc1eecac8 100644 --- a/src/coreclr/tools/Common/CommandLineHelpers.cs +++ b/src/coreclr/tools/Common/CommandLineHelpers.cs @@ -27,6 +27,8 @@ internal static partial class Helpers public static string[] ValidOS { get; } = ["windows", "linux", "freebsd", "openbsd", "osx", "maccatalyst", "ios", "iossimulator", "tvos", "tvossimulator", "android", "browser", "wasi"]; public static string[] ValidArchitectures { get; } = ["arm", "armel", "arm64", "x86", "x64", "riscv64", "loongarch64", "wasm"]; + // Targets that only the NativeAOT compiler supports (no ReadyToRun): the RISC-V lp64 soft-float ABI. + public static string[] ValidArchitecturesNativeAot { get; } = [.. ValidArchitectures, "riscv64-lp64"]; public static Dictionary BuildPathDictionary(IReadOnlyList tokens, bool strict) { @@ -119,7 +121,7 @@ public static TargetArchitecture GetTargetArchitecture(string token) "arm64" => TargetArchitecture.ARM64, "wasm" => TargetArchitecture.Wasm32, "loongarch64" => TargetArchitecture.LoongArch64, - "riscv64" => TargetArchitecture.RiscV64, + "riscv64" or "riscv64-lp64" => TargetArchitecture.RiscV64, _ => throw new CommandLineException($"Target architecture '{token}' is not supported") }; } @@ -136,6 +138,7 @@ public static (TargetArchitecture, TargetOS, TargetAbi) GetTargetSpec(string tar { (_, "armel") => TargetAbi.NativeAotArmel, ("android", "arm") => TargetAbi.NativeAotArmel, + (_, "riscv64-lp64") => TargetAbi.NativeAotRiscV64SoftFloat, _ => TargetAbi.NativeAot, }; diff --git a/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs b/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs index 3e252c360be5ea..ad48f024bc5e54 100644 --- a/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs +++ b/src/coreclr/tools/Common/Compiler/ObjectWriter/ElfObjectWriter.cs @@ -41,6 +41,7 @@ internal sealed partial class ElfObjectWriter : UnixObjectWriter private readonly bool _useInlineRelocationAddends; private readonly ushort _machine; private readonly bool _useSoftFPAbi; + private readonly uint _riscV64ElfFlags; private readonly List _sections = new(); private readonly List _symbols = new(); private uint _localSymbolCount; @@ -53,6 +54,25 @@ internal sealed partial class ElfObjectWriter : UnixObjectWriter private static readonly ObjectNodeSection ArmTextThunkSection = new ObjectNodeSection(".text.thunks", SectionType.Executable); private static readonly ObjectNodeSection CommentSection = new ObjectNodeSection(".comment", SectionType.ReadOnly); + /// + /// RISC-V ELF header flags: the floating-point ABI comes from the target ABI + /// (EF_RISCV_FLOAT_ABI_DOUBLE for lp64d, EF_RISCV_FLOAT_ABI_SOFT for lp64 - the + /// linker rejects objects with mixed floating-point ABIs), EF_RISCV_RVC is set + /// when the target has the C extension. + /// + internal static uint GetRiscV64ElfFlags(TargetAbi abi, ObjectWritingOptions options) + { + const uint EF_RISCV_RVC = 0x0001; + const uint EF_RISCV_FLOAT_ABI_DOUBLE = 0x0004; + + uint flags = abi == TargetAbi.NativeAotRiscV64SoftFloat ? 0u : EF_RISCV_FLOAT_ABI_DOUBLE; + if (options.HasFlag(ObjectWritingOptions.RiscV64Compressed)) + { + flags |= EF_RISCV_RVC; + } + return flags; + } + public ElfObjectWriter(NodeFactory factory, ObjectWritingOptions options) : base(factory, options) { @@ -68,6 +88,7 @@ public ElfObjectWriter(NodeFactory factory, ObjectWritingOptions options) }; _useInlineRelocationAddends = _machine is EM_386 or EM_ARM; _useSoftFPAbi = _machine is EM_ARM && factory.Target.Abi == TargetAbi.NativeAotArmel; + _riscV64ElfFlags = GetRiscV64ElfFlags(factory.Target.Abi, options); // By convention the symbol table starts with empty symbol _symbols.Add(new ElfSymbol {}); @@ -765,7 +786,7 @@ private void EmitObjectFile(Stream outputFileStream) { EM_ARM => 0x05000000u, // For ARM32 claim conformance with the EABI specification EM_LOONGARCH => 0x43u, // For LoongArch ELF psABI specify the ABI version (1) and modifiers (64-bit GPRs, 64-bit FPRs) - EM_RISCV => 0x0005u, // EF_RISCV_RVC (RVC ABI) | EF_RISCV_FLOAT_ABI_DOUBLE (double precision floating-point ABI). + EM_RISCV => _riscV64ElfFlags, _ => 0u }, }; diff --git a/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs b/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs index 61276810208aca..7e9f4e61f197a5 100644 --- a/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs +++ b/src/coreclr/tools/Common/Compiler/ObjectWriter/ObjectWritingOptions.cs @@ -13,5 +13,9 @@ public enum ObjectWritingOptions ControlFlowGuard = 0x02, UseDwarf5 = 0x4, GenerateUnwindInfo = 0x8, + /// + /// RISC-V: the target supports the C (compressed instructions) extension + /// + RiscV64Compressed = 0x10, } } diff --git a/src/coreclr/tools/Common/InstructionSetHelpers.cs b/src/coreclr/tools/Common/InstructionSetHelpers.cs index fd256c194bb01b..7c4d16241d3e9f 100644 --- a/src/coreclr/tools/Common/InstructionSetHelpers.cs +++ b/src/coreclr/tools/Common/InstructionSetHelpers.cs @@ -18,7 +18,7 @@ namespace System.CommandLine internal static partial class Helpers { public static InstructionSetSupport ConfigureInstructionSetSupport(string instructionSet, int maxVectorTBitWidth, bool isVectorTOptimistic, TargetArchitecture targetArchitecture, TargetOS targetOS, - string mustNotBeMessage, string invalidImplicationMessage, Logger logger, bool allowOptimistic, bool isReadyToRun) + string mustNotBeMessage, string invalidImplicationMessage, Logger logger, bool allowOptimistic, bool isReadyToRun, TargetAbi targetAbi) { InstructionSetSupportBuilder instructionSetSupportBuilder = new(targetArchitecture); @@ -98,9 +98,13 @@ public static InstructionSetSupport ConfigureInstructionSetSupport(string instru { // The rv64gc baseline: D implies F, so "d", "c" and "a" cover // the G+C extensions. Reduced-ISA targets (e.g. zkVM guests) - // opt out with --instruction-set=-a,-c,-d,-f. + // opt out with --instruction-set=-a,-c,-d,-f. The lp64 (soft-float) + // ABI target has no F/D by definition. instructionSetSupportBuilder.AddSupportedInstructionSet("base"); - instructionSetSupportBuilder.AddSupportedInstructionSet("d"); + if (targetAbi != TargetAbi.NativeAotRiscV64SoftFloat) + { + instructionSetSupportBuilder.AddSupportedInstructionSet("d"); + } instructionSetSupportBuilder.AddSupportedInstructionSet("c"); instructionSetSupportBuilder.AddSupportedInstructionSet("a"); } diff --git a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs index 5ff98b8f980021..5f21a5c8866761 100644 --- a/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs +++ b/src/coreclr/tools/Common/Internal/Runtime/ReadyToRunConstants.cs @@ -409,6 +409,22 @@ public enum ReadyToRunHelper TypeHandleToRuntimeType, GetRefAny, TypeHandleToRuntimeTypeHandle, + + // Soft-float arithmetic (targets without an FPU). NativeAOT only. + FltAdd, + FltSub, + FltMul, + FltDiv, + DblAdd, + DblSub, + DblMul, + DblDiv, + FltCmpLe, + FltCmpGe, + DblCmpLe, + DblCmpGe, + Flt2Dbl, + Dbl2Flt, } // Enum used for HFA type recognition. diff --git a/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs b/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs index d874da3451ae71..6eddb38bcdb66d 100644 --- a/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs +++ b/src/coreclr/tools/Common/JitInterface/CorInfoHelpFunc.cs @@ -41,6 +41,22 @@ public enum CorInfoHelpFunc CORINFO_HELP_FLTREM, CORINFO_HELP_DBLREM, + // Soft-float helpers (targets without an FPU); see corinfo.h + CORINFO_HELP_FLTADD, + CORINFO_HELP_FLTSUB, + CORINFO_HELP_FLTMUL, + CORINFO_HELP_FLTDIV, + CORINFO_HELP_DBLADD, + CORINFO_HELP_DBLSUB, + CORINFO_HELP_DBLMUL, + CORINFO_HELP_DBLDIV, + CORINFO_HELP_FLTCMP_LE, + CORINFO_HELP_FLTCMP_GE, + CORINFO_HELP_DBLCMP_LE, + CORINFO_HELP_DBLCMP_GE, + CORINFO_HELP_FLT2DBL, + CORINFO_HELP_DBL2FLT, + /* Allocating a new object. Always use ICorClassInfo::getNewHelper() to decide which is the right helper to use to allocate an object of a given type. */ diff --git a/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs b/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs index 615f5c91ccf897..0048f8a4abadd2 100644 --- a/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs +++ b/src/coreclr/tools/Common/JitInterface/CorInfoImpl.cs @@ -4915,6 +4915,13 @@ private uint getJitFlags(ref CORJIT_FLAGS flags, uint sizeInBytes) flags.Set(CorJitFlag.CORJIT_FLAG_SOFTFP_ABI); } + if (this.MethodBeingCompiled.Context.Target.Abi == TargetAbi.NativeAotRiscV64SoftFloat) + { + // RISC-V lp64: FP values are passed in integer registers and the FP + // arithmetic goes through the soft-float helpers. + flags.Set(CorJitFlag.CORJIT_FLAG_SOFTFP_ABI); + } + if (this.MethodBeingCompiled.IsAsyncCall() #if !READYTORUN || (_compilation.TypeSystemContext.IsSpecialUnboxingThunk(this.MethodBeingCompiled) && _compilation.TypeSystemContext.GetTargetOfSpecialUnboxingThunk(this.MethodBeingCompiled).IsAsyncCall()) diff --git a/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs b/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs index 4c9862cd0c8698..2ceac0f2363a91 100644 --- a/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs +++ b/src/coreclr/tools/Common/TypeSystem/Common/TargetDetails.cs @@ -39,6 +39,11 @@ public enum TargetAbi /// model for armel execution model /// NativeAotArmel, + /// + /// RISC-V lp64 (soft-float) execution model: no F/D extensions, floating-point + /// values are passed in integer registers and computed by the soft-float helpers + /// + NativeAotRiscV64SoftFloat, } /// diff --git a/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs b/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs index 99fc77b6145ac3..455f30df1e021c 100644 --- a/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs +++ b/src/coreclr/tools/aot/ILCompiler.Compiler/Compiler/JitHelper.cs @@ -219,6 +219,50 @@ public static void GetEntryPoint(TypeSystemContext context, ReadyToRunHelper id, mangledName = "fmodf"; break; + // Soft-float arithmetic: the compiler-rt/libgcc builtins of the target toolchain + case ReadyToRunHelper.FltAdd: + mangledName = "__addsf3"; + break; + case ReadyToRunHelper.FltSub: + mangledName = "__subsf3"; + break; + case ReadyToRunHelper.FltMul: + mangledName = "__mulsf3"; + break; + case ReadyToRunHelper.FltDiv: + mangledName = "__divsf3"; + break; + case ReadyToRunHelper.DblAdd: + mangledName = "__adddf3"; + break; + case ReadyToRunHelper.DblSub: + mangledName = "__subdf3"; + break; + case ReadyToRunHelper.DblMul: + mangledName = "__muldf3"; + break; + case ReadyToRunHelper.DblDiv: + mangledName = "__divdf3"; + break; + case ReadyToRunHelper.FltCmpLe: + mangledName = "__lesf2"; + break; + case ReadyToRunHelper.FltCmpGe: + mangledName = "__gesf2"; + break; + case ReadyToRunHelper.DblCmpLe: + mangledName = "__ledf2"; + break; + case ReadyToRunHelper.DblCmpGe: + mangledName = "__gedf2"; + break; + case ReadyToRunHelper.Flt2Dbl: + mangledName = "__extendsfdf2"; + break; + case ReadyToRunHelper.Dbl2Flt: + mangledName = "__truncdfsf2"; + break; + case ReadyToRunHelper.LMul: mangledName = "RhpLMul"; break; diff --git a/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs b/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs index 86fdf190fd18cd..eb0d7b924b1030 100644 --- a/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs +++ b/src/coreclr/tools/aot/ILCompiler.Diagnostics/PerfMapWriter.cs @@ -128,6 +128,7 @@ private static PerfmapTokensForTarget TranslateTargetDetailsToPerfmapConstants(T TargetAbi.Unknown => PerfMapAbiToken.Unknown, TargetAbi.NativeAot => PerfMapAbiToken.Default, TargetAbi.NativeAotArmel => PerfMapAbiToken.Armel, + TargetAbi.NativeAotRiscV64SoftFloat => PerfMapAbiToken.Default, _ => throw new NotImplementedException(details.Abi.ToString()) }; diff --git a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs index 827907a5aa43f3..1182849b6b3fd5 100644 --- a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs +++ b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCases/R2RTestSuites.cs @@ -66,6 +66,29 @@ static void Validate(ReadyToRunReader reader) } } + [Fact] + public void RiscV64SoftFloatTargetIsRejected() + { + // The riscv64-lp64 (soft-float) ABI is NativeAOT-only: crossgen2 rejects it on the + // command line instead of compiling with helpers that have no ReadyToRun encoding. + var module = new CompiledAssembly + { + AssemblyName = nameof(RiscV64SoftFloatTargetIsRejected), + SourceResourceNames = ["ThumbBit/HotColdSplitting.cs"], + }; + + new R2RTestRunner(_output).Run(new R2RTestCase( + nameof(RiscV64SoftFloatTargetIsRejected), + [ + new(nameof(RiscV64SoftFloatTargetIsRejected), [new CrossgenAssembly(module)]) + { + TargetOS = "linux", + TargetArchitecture = "riscv64-lp64", + ExpectedFailure = "is not supported by ReadyToRun", + }, + ])); + } + [Fact] public void GenericTypeConstraintsAllowVariantParameters() { diff --git a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs index 4fdb38bc20efbf..11b18769ae942b 100644 --- a/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs +++ b/src/coreclr/tools/aot/ILCompiler.ReadyToRun.Tests/TestCasesRunner/R2RTestRunner.cs @@ -94,6 +94,21 @@ internal sealed class CrossgenCompilation(string name, List as /// public Action? Validate { get; init; } + /// + /// When set, crossgen2 is expected to fail and its standard error must contain this text. + /// + public string? ExpectedFailure { get; init; } + + /// + /// Overrides the target OS passed to crossgen2; defaults to the test run's target. + /// + public string? TargetOS { get; init; } + + /// + /// Overrides the target architecture passed to crossgen2; defaults to the test run's target. + /// + public string? TargetArchitecture { get; init; } + public string Name => name; public bool IsComposite => Options.Contains(Crossgen2Option.Composite); @@ -309,7 +324,7 @@ private static string RunCrossgenCompilation( foreach (var option in compilation.Options) args.Add(option.ToArg()); - args.AddRange(["--targetos", TestPaths.TargetOS, "--targetarch", TestPaths.TargetArchitecture]); + args.AddRange(["--targetos", compilation.TargetOS ?? TestPaths.TargetOS, "--targetarch", compilation.TargetArchitecture ?? TestPaths.TargetArchitecture]); // Caller-supplied raw args (for options that take values, e.g. --determinism-stress=N) args.AddRange(compilation.AdditionalArgs); @@ -324,6 +339,12 @@ private static string RunCrossgenCompilation( args.Add($"--out"); args.Add($"{outputFile}"); var result = driver.Compile(args); + if (compilation.ExpectedFailure is string expectedFailure) + { + Assert.False(result.Success, $"crossgen2 unexpectedly succeeded for '{testName}'"); + Assert.Contains(expectedFailure, result.StandardError); + return outputFile; + } Assert.True(result.Success, $"crossgen2 failed for '{testName}':\n{result.StandardError}\n{result.StandardOutput}"); diff --git a/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs b/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs index e812ae60c3e2f3..9c3e87178be27b 100644 --- a/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs +++ b/src/coreclr/tools/aot/ILCompiler.RyuJit/Compiler/RyuJitCompilation.cs @@ -60,6 +60,26 @@ internal RyuJitCompilation( _compilationOptions = options; InstructionSetSupport = instructionSetSupport; + if (nodeFactory.Target.Architecture == TargetArchitecture.RiscV64) + { + // The ABI is a property of the target, not of the instruction set: opting out of + // F/D does not make the runtime and the native libraries lp64, and the lp64 + // target has no F/D by definition. The lp64f ABI (F without D) is not supported. + bool hasF = instructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_F); + bool hasD = instructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_D); + if (nodeFactory.Target.Abi == TargetAbi.NativeAotRiscV64SoftFloat) + { + if (hasF || hasD) + throw new NotSupportedException( + "The riscv64-lp64 (soft-float) ABI target has no F and D extensions; remove them from the instruction set."); + } + else if (!hasF || !hasD) + { + throw new NotSupportedException( + "The riscv64 lp64d ABI requires the F and D extensions; targets without them must use the riscv64-lp64 (soft-float) ABI."); + } + } + _profileDataManager = profileDataManager; _methodImportationErrorProvider = errorProvider; @@ -126,6 +146,10 @@ protected override void CompileInternal(string outputFile, ObjectDumper dumper) if ((_compilationOptions & RyuJitCompilationOptions.ControlFlowGuardAnnotations) != 0) options |= ObjectWritingOptions.ControlFlowGuard; + if ((NodeFactory.Target.Architecture == TargetArchitecture.RiscV64) + && InstructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_C)) + options |= ObjectWritingOptions.RiscV64Compressed; + ObjectWriter.ObjectWriter.EmitObject(outputFile, nodes, NodeFactory, options, dumper, _logger); } diff --git a/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs b/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs index c17dae1393b11f..f5cc12cf61694c 100644 --- a/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs +++ b/src/coreclr/tools/aot/ILCompiler.RyuJit/JitInterface/CorInfoImpl.RyuJit.cs @@ -737,6 +737,48 @@ private ISymbolNode GetHelperFtnUncached(CorInfoHelpFunc ftnNum) case CorInfoHelpFunc.CORINFO_HELP_DBLREM: id = ReadyToRunHelper.DblRem; break; + case CorInfoHelpFunc.CORINFO_HELP_FLTADD: + id = ReadyToRunHelper.FltAdd; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTSUB: + id = ReadyToRunHelper.FltSub; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTMUL: + id = ReadyToRunHelper.FltMul; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTDIV: + id = ReadyToRunHelper.FltDiv; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLADD: + id = ReadyToRunHelper.DblAdd; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLSUB: + id = ReadyToRunHelper.DblSub; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLMUL: + id = ReadyToRunHelper.DblMul; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLDIV: + id = ReadyToRunHelper.DblDiv; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTCMP_LE: + id = ReadyToRunHelper.FltCmpLe; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLTCMP_GE: + id = ReadyToRunHelper.FltCmpGe; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLCMP_LE: + id = ReadyToRunHelper.DblCmpLe; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBLCMP_GE: + id = ReadyToRunHelper.DblCmpGe; + break; + case CorInfoHelpFunc.CORINFO_HELP_FLT2DBL: + id = ReadyToRunHelper.Flt2Dbl; + break; + case CorInfoHelpFunc.CORINFO_HELP_DBL2FLT: + id = ReadyToRunHelper.Dbl2Flt; + break; case CorInfoHelpFunc.CORINFO_HELP_JIT_PINVOKE_BEGIN: id = ReadyToRunHelper.PInvokeBegin; diff --git a/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs b/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs index 875bbeaf73b9e7..9bed4be8a61018 100644 --- a/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs +++ b/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs @@ -352,7 +352,7 @@ public static void PrintExtendedHelp(ParseResult _) Console.WriteLine("Valid switches for {0} are: '{1}'. The default value is '{2}'\n", "--targetos", string.Join("', '", Helpers.ValidOS), Helpers.GetTargetOS(null).ToString().ToLowerInvariant()); - Console.WriteLine(string.Format("Valid switches for {0} are: '{1}'. The default value is '{2}'\n", "--targetarch", string.Join("', '", Helpers.ValidArchitectures), Helpers.GetTargetArchitecture(null).ToString().ToLowerInvariant())); + Console.WriteLine(string.Format("Valid switches for {0} are: '{1}'. The default value is '{2}'\n", "--targetarch", string.Join("', '", Helpers.ValidArchitecturesNativeAot), Helpers.GetTargetArchitecture(null).ToString().ToLowerInvariant())); Console.WriteLine("The allowable values for the --instruction-set option are described in the table below. Each architecture has a different set of valid " + "instruction sets, and multiple instruction sets may be specified by separating the instructions sets by a ','. By default other instruction sets not " + @@ -360,7 +360,7 @@ public static void PrintExtendedHelp(ParseResult _) "All such light-up can be disallowed by specifying '-optimistic'. The instruction sets supported by the machine invoking the tool can be targeted by " + "specifying 'native'. For example 'native', 'avx,aes', 'avx,aes,-avx2', or 'avx,aes,-optimistic'"); - foreach (string arch in Helpers.ValidArchitectures) + foreach (string arch in Helpers.ValidArchitecturesNativeAot) { TargetArchitecture targetArch = Helpers.GetTargetArchitecture(arch); bool first = true; diff --git a/src/coreclr/tools/aot/ILCompiler/Program.cs b/src/coreclr/tools/aot/ILCompiler/Program.cs index 34305bc9ebf504..86aac8ced8393d 100644 --- a/src/coreclr/tools/aot/ILCompiler/Program.cs +++ b/src/coreclr/tools/aot/ILCompiler/Program.cs @@ -109,7 +109,8 @@ public int Run() InstructionSetSupport instructionSetSupport = Helpers.ConfigureInstructionSetSupport(Get(_command.InstructionSet), Get(_command.MaxVectorTBitWidth), isVectorTOptimistic, targetArchitecture, targetOS, "Unrecognized instruction set {0}", "Unsupported combination of instruction sets: {0}/{1}", logger, allowOptimistic: _command.OptimizationMode != OptimizationMode.PreferSize, - isReadyToRun: false); + isReadyToRun: false, + targetAbi: targetAbi); string systemModuleName = Get(_command.SystemModuleName); string reflectionData = Get(_command.ReflectionData); diff --git a/src/coreclr/tools/aot/crossgen2/Program.cs b/src/coreclr/tools/aot/crossgen2/Program.cs index ca0c40ac43b632..cb6203c5d0539f 100644 --- a/src/coreclr/tools/aot/crossgen2/Program.cs +++ b/src/coreclr/tools/aot/crossgen2/Program.cs @@ -80,6 +80,11 @@ public int Run() (TargetArchitecture targetArchitecture, TargetOS targetOS, TargetAbi targetAbi) = Helpers.GetTargetSpec(Get(_command.TargetArchitecture), Get(_command.TargetOS)); + if (targetAbi == TargetAbi.NativeAotRiscV64SoftFloat) + { + // The soft-float helpers have no ReadyToRun encoding; the target is NativeAOT only. + throw new CommandLineException($"Target architecture '{Get(_command.TargetArchitecture)}' is not supported by ReadyToRun"); + } // The portable call-helpers generator is currently supported only for Wasm. if (_generatePortableCallHelpers is not null @@ -108,7 +113,8 @@ public int Run() InstructionSetSupport instructionSetSupport = Helpers.ConfigureInstructionSetSupport(Get(_command.InstructionSet), Get(_command.MaxVectorTBitWidth), isVectorTOptimistic, targetArchitecture, targetOS, SR.InstructionSetMustNotBe, SR.InstructionSetInvalidImplication, logger, allowOptimistic: allowOptimistic, - isReadyToRun: true); + isReadyToRun: true, + targetAbi: targetAbi); if (!targetAllowsRuntimeCodeGeneration) { instructionSetSupport = Helpers.GetFixedInstructionSetSupport(instructionSetSupport); From 99ea61bab5e47a3848e2f776a5c36777ef3d5fb2 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 9 Sep 2026 11:27:25 +0100 Subject: [PATCH 10/16] [RISC-V] Generate code for the soft-float target The IR keeps TYP_FLOAT and TYP_DOUBLE; what changes under the soft-float ABI is their register class, which becomes VTR_INT. Most of the backend already decides by register class, so FP values are then allocated, spilled, moved, passed and returned in integer registers with no further work; the places that decide by varTypeIsFloating instead are adjusted. The operations are expanded in global morph through the existing hooks: arithmetic and conversions into helper calls, comparisons into a compare helper plus an integer relop, negation into a sign-bit XOR, and CKFINITE into a bounds check on the exponent field. FP Math intrinsics are declined by type and remain calls to the managed implementations. The debug JIT asserts on any F/D opcode reaching the emitter when F is not in the instruction set, so an unexpanded node fails on the method that produced it. Signed-off-by: Maxim Menshikov --- src/coreclr/jit/codegencommon.cpp | 9 + src/coreclr/jit/codegenriscv64.cpp | 20 + src/coreclr/jit/compiler.cpp | 54 ++- src/coreclr/jit/compiler.h | 9 + src/coreclr/jit/emitriscv64.cpp | 2 +- src/coreclr/jit/flowgraph.cpp | 9 +- src/coreclr/jit/importer.cpp | 2 +- src/coreclr/jit/importercalls.cpp | 12 +- src/coreclr/jit/instr.cpp | 7 +- src/coreclr/jit/jit.h | 6 + src/coreclr/jit/lclvars.cpp | 7 +- src/coreclr/jit/lower.cpp | 5 +- src/coreclr/jit/lsra.cpp | 8 + src/coreclr/jit/lsrabuild.cpp | 11 +- src/coreclr/jit/lsrariscv64.cpp | 6 +- src/coreclr/jit/morph.cpp | 447 ++++++++++++++++++ src/coreclr/jit/utils.cpp | 4 + src/coreclr/jit/vartype.h | 10 + src/coreclr/nativeaot/Runtime/MathHelpers.cpp | 18 +- 19 files changed, 622 insertions(+), 24 deletions(-) diff --git a/src/coreclr/jit/codegencommon.cpp b/src/coreclr/jit/codegencommon.cpp index 8ea9c5f49c7850..29ca4d2fe39bde 100644 --- a/src/coreclr/jit/codegencommon.cpp +++ b/src/coreclr/jit/codegencommon.cpp @@ -8571,6 +8571,15 @@ void CodeGen::genPoisonFrame(regMaskTP regLiveIn) // void CodeGen::genBitCast(var_types targetType, regNumber targetReg, var_types srcType, regNumber srcReg) { +#ifdef TARGET_RISCV64 + if (m_compiler->opts.compUseSoftFP && (srcType == TYP_FLOAT) && (targetType == TYP_INT)) + { + // Soft-float: a float lives in an integer register with unspecified upper + // bits (RISC-V psABI), while an int is expected to be sign-extended. + GetEmitter()->emitIns_R_R_I(INS_addiw, EA_4BYTE, targetReg, srcReg, 0); + return; + } +#endif // TARGET_RISCV64 const bool srcFltReg = varTypeUsesFloatReg(srcType); assert(srcFltReg == genIsValidFloatReg(srcReg)); diff --git a/src/coreclr/jit/codegenriscv64.cpp b/src/coreclr/jit/codegenriscv64.cpp index aa4abea743baca..9ee86fa9ab0ab1 100644 --- a/src/coreclr/jit/codegenriscv64.cpp +++ b/src/coreclr/jit/codegenriscv64.cpp @@ -1008,6 +1008,26 @@ void CodeGen::genSetRegToConst(regNumber targetReg, var_types targetType, GenTre emitAttr size = emitActualTypeSize(tree); double constValue = tree->AsDblCon()->DconValue(); + if (m_compiler->opts.compUseSoftFP) + { + // No FP registers: the constant is its IEEE 754 bit pattern in an integer register. + assert(genIsValidIntReg(targetReg)); + int64_t bits; + if (size == EA_4BYTE) + { + float fltValue = (float)constValue; + int32_t fltBits; + memcpy(&fltBits, &fltValue, sizeof(fltBits)); + bits = fltBits; + } + else + { + memcpy(&bits, &constValue, sizeof(bits)); + } + instGen_Set_Reg_To_Imm(size, targetReg, bits); + break; + } + assert(emitter::isFloatReg(targetReg)); int64_t bits; if (emitter::isSingleInstructionFpImm(constValue, size, &bits)) diff --git a/src/coreclr/jit/compiler.cpp b/src/coreclr/jit/compiler.cpp index c907297271cb46..45c6f6ecff3f9a 100644 --- a/src/coreclr/jit/compiler.cpp +++ b/src/coreclr/jit/compiler.cpp @@ -53,6 +53,9 @@ MethodSet* Compiler::s_pJitMethodSet = nullptr; bool GlobalJitOptions::compFeatureHfa = false; LONG GlobalJitOptions::compUseSoftFPConfigured = 0; #endif // CONFIGURABLE_ARM_ABI +#ifdef TARGET_RISCV64 +LONG GlobalJitOptions::compUseSoftFPConfigured = 0; +#endif // TARGET_RISCV64 /***************************************************************************** * @@ -2502,6 +2505,11 @@ void Compiler::compInitOptions(JitFlags* jitFlags) if (compIsForInlining()) { +#ifdef TARGET_RISCV64 + // The soft-float mode is decided by the root compilation (see below); the + // importer of an inlinee needs it too, for the intrinsics and casts it expands. + opts.compUseSoftFP = impInlineInfo->InlinerCompiler->opts.compUseSoftFP; +#endif // TARGET_RISCV64 return; } @@ -2907,10 +2915,50 @@ void Compiler::compInitOptions(JitFlags* jitFlags) GlobalJitOptions::compFeatureHfa = !opts.compUseSoftFP; #elif defined(TARGET_RISCV64) - // lp64 soft-float ABI: FP scalars and FP struct fields are passed by the - // integer calling convention and fa* registers are never used. Set by the - // VM / AOT driver for targets without the F extension. + // Soft-float, for targets without the F/D extensions (set by the AOT driver): + // the lp64 calling convention passes FP values in integer registers, and + // TYP_FLOAT/TYP_DOUBLE values live in the integer register file altogether; + // the FP arithmetic is done by helper calls (see fgMorphSmpOp). The register + // class of a type is a process-wide table, so the setting cannot change + // during the lifetime of the process. opts.compUseSoftFP = jitFlags->IsSet(JitFlags::JIT_FLAG_SOFTFP_ABI); + + // The first compilation of the process fixes the mode: it claims the + // configuration, initializes the table and then publishes the mode. Every + // other compilation waits for the publication and must request the same + // mode, so the table is never written while another compilation may read it. + enum SoftFPConfig : LONG + { + SoftFPConfigUnset = 0, + SoftFPConfigHard = 1, + SoftFPConfigSoft = 2, + SoftFPConfigInitializing = 3, + }; + const LONG softFPConfig = opts.compUseSoftFP ? SoftFPConfigSoft : SoftFPConfigHard; + LONG oldSoftFPConfig = InterlockedCompareExchange(&GlobalJitOptions::compUseSoftFPConfigured, + SoftFPConfigInitializing, SoftFPConfigUnset); + if (oldSoftFPConfig == SoftFPConfigUnset) + { + if (opts.compUseSoftFP) + { + varTypeRegister[TYP_FLOAT] = VTR_INT; + varTypeRegister[TYP_DOUBLE] = VTR_INT; + } + InterlockedExchange(&GlobalJitOptions::compUseSoftFPConfigured, softFPConfig); + } + else + { + while (oldSoftFPConfig == SoftFPConfigInitializing) + { + // Atomic read; the initialization window is two byte stores long. + oldSoftFPConfig = InterlockedCompareExchange(&GlobalJitOptions::compUseSoftFPConfigured, SoftFPConfigUnset, + SoftFPConfigUnset); + } + if (oldSoftFPConfig != softFPConfig) + { + NO_WAY("SoftFP setting changed during lifetime of process"); + } + } #elif defined(ARM_SOFTFP) && defined(TARGET_ARM) // Armel is unconditionally enabled in the JIT. Verify that the VM side agrees. assert(jitFlags->IsSet(JitFlags::JIT_FLAG_SOFTFP_ABI)); diff --git a/src/coreclr/jit/compiler.h b/src/coreclr/jit/compiler.h index a355f05a402625..ec793af1368e7f 100644 --- a/src/coreclr/jit/compiler.h +++ b/src/coreclr/jit/compiler.h @@ -7285,6 +7285,15 @@ class Compiler bool fgIsBlockCold(BasicBlock* block); GenTree* fgMorphCastIntoHelper(GenTree* tree, int helper, GenTree* oper); +#ifdef TARGET_RISCV64 + // Soft-float expansion of the FP operations into helper calls + GenTree* fgMorphSoftFloatArith(GenTreeOp* tree); + GenTree* fgMorphSoftFloatCast(GenTreeCast* tree, CorInfoHelpFunc helper, GenTree* oper); + GenTree* fgMorphSoftFloatNeg(GenTreeOp* neg); + GenTree* fgMorphSoftFloatRelop(GenTreeOp* relop); + GenTree* fgMorphSoftFloatCkFinite(GenTreeOp* ckFinite); + GenTree* fgMorphSoftFloatCastToInt32(GenTree* src, bool toUnsigned); +#endif // TARGET_RISCV64 GenTree* fgMorphIntoHelperCall( GenTree* tree, int helper, bool morphArgs, GenTree* arg1 = nullptr, GenTree* arg2 = nullptr); diff --git a/src/coreclr/jit/emitriscv64.cpp b/src/coreclr/jit/emitriscv64.cpp index 02c4ddb5f41ac3..3620f01a4c30fc 100644 --- a/src/coreclr/jit/emitriscv64.cpp +++ b/src/coreclr/jit/emitriscv64.cpp @@ -4994,7 +4994,7 @@ void emitter::emitInsLoadStoreOp(instruction ins, emitAttr attr, regNumber dataR } else { - bool needTemp = indir->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeIsFloating(indir); + bool needTemp = indir->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeUsesFloatReg(indir); if (addr->AsIntCon()->FitsInAddrBase(m_compiler) && addr->AsIntCon()->AddrNeedsReloc(m_compiler)) { regNumber addrReg = needTemp ? codeGen->internalRegisters.GetSingle(indir) : dataReg; diff --git a/src/coreclr/jit/flowgraph.cpp b/src/coreclr/jit/flowgraph.cpp index b8f1e8d05c999f..842a51c39c7edb 100644 --- a/src/coreclr/jit/flowgraph.cpp +++ b/src/coreclr/jit/flowgraph.cpp @@ -1359,6 +1359,13 @@ bool Compiler::fgCastRequiresHelper(var_types fromType, var_types toType, bool o #endif // TARGET_X86 } #endif // TARGET_X86 || TARGET_ARM +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && (varTypeIsFloating(fromType) || varTypeIsFloating(toType))) + { + // No FP instructions: every conversion to or from floating point is a helper call. + return true; + } +#endif // TARGET_RISCV64 return false; } @@ -2171,7 +2178,7 @@ class MergedReturns retVarDsc->lvType = retLclType; } - if (varTypeIsFloating(retVarDsc->TypeGet())) + if (varTypeIsFloating(retVarDsc->TypeGet()) && varTypeUsesFloatReg(retVarDsc->TypeGet())) { m_compiler->compFloatingPointUsed = true; } diff --git a/src/coreclr/jit/importer.cpp b/src/coreclr/jit/importer.cpp index fe8dd181aab974..89f2fc9309bc41 100644 --- a/src/coreclr/jit/importer.cpp +++ b/src/coreclr/jit/importer.cpp @@ -41,7 +41,7 @@ void Compiler::impPushOnStack(GenTree* tree, typeInfo ti) { compLongUsed = true; } - else if (tree->TypeIs(TYP_FLOAT) || tree->TypeIs(TYP_DOUBLE)) + else if (varTypeIsFloating(tree) && varTypeUsesFloatReg(tree)) { compFloatingPointUsed = true; } diff --git a/src/coreclr/jit/importercalls.cpp b/src/coreclr/jit/importercalls.cpp index cd8c8441aaf4fe..147bccf30c25aa 100644 --- a/src/coreclr/jit/importercalls.cpp +++ b/src/coreclr/jit/importercalls.cpp @@ -5846,7 +5846,8 @@ GenTree* Compiler::impIntrinsic(CORINFO_CLASS_HANDLE clsHnd, #endif // FEATURE_HW_INTRINSICS #ifdef TARGET_RISCV64 - if (!isMagnitude) + // Soft-float: no fmin/fmax, the managed implementation is called instead. + if (!isMagnitude && !opts.compUseSoftFP) { GenTree* op2 = impImplicitR4orR8Cast(impPopStack().val, callType); GenTree* op1 = impImplicitR4orR8Cast(impPopStack().val, callType); @@ -11492,6 +11493,15 @@ GenTree* Compiler::impMathIntrinsic(CORINFO_METHOD_HANDLE method, assert(IsMathIntrinsic(intrinsicName)); assert(isSpecial != nullptr); +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(callType)) + { + // No FP instructions: leave the call to the managed implementation. + JITDUMP("Soft-float: math intrinsic %d is left as a call\n", (int)intrinsicName); + return nullptr; + } +#endif // TARGET_RISCV64 + op1 = nullptr; bool isIntrinsicImplementedByUserCall = IsIntrinsicImplementedByUserCall(intrinsicName); diff --git a/src/coreclr/jit/instr.cpp b/src/coreclr/jit/instr.cpp index cb5c3396233539..9ba41860724fbf 100644 --- a/src/coreclr/jit/instr.cpp +++ b/src/coreclr/jit/instr.cpp @@ -2156,8 +2156,9 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false* else ins = INS_lh; } - else if (TYP_INT == srcType) + else if ((TYP_INT == srcType) || (TYP_FLOAT == srcType)) { + // TYP_FLOAT: soft-float, the value lives in an integer register. ins = INS_lw; } else @@ -2521,8 +2522,8 @@ instruction CodeGenInterface::ins_Store(var_types dstType, bool aligned /*=false ins = INS_sb; else if (varTypeIsShort(dstType)) ins = INS_sh; - else if (TYP_INT == dstType) - ins = INS_sw; + else if ((TYP_INT == dstType) || (TYP_FLOAT == dstType)) + ins = INS_sw; // TYP_FLOAT: soft-float, the value lives in an integer register. else ins = INS_sd; #else diff --git a/src/coreclr/jit/jit.h b/src/coreclr/jit/jit.h index 376fc80e62f661..7fa7a798b5f96d 100644 --- a/src/coreclr/jit/jit.h +++ b/src/coreclr/jit/jit.h @@ -476,6 +476,12 @@ class GlobalJitOptions #else // !FEATURE_HFA static const bool compFeatureHfa = false; #endif // FEATURE_HFA +#ifdef TARGET_RISCV64 + // Soft-float changes the register class of TYP_FLOAT/TYP_DOUBLE (varTypeRegister), + // which is a process-wide table; the mode is fixed by the first compilation + // (see compInitOptions for the states). + static LONG compUseSoftFPConfigured; +#endif // TARGET_RISCV64 #ifdef FEATURE_HFA #undef FEATURE_HFA diff --git a/src/coreclr/jit/lclvars.cpp b/src/coreclr/jit/lclvars.cpp index 0ee7b7042aa4ad..6ba00046a5a3f5 100644 --- a/src/coreclr/jit/lclvars.cpp +++ b/src/coreclr/jit/lclvars.cpp @@ -676,7 +676,10 @@ void Compiler::lvaInitUserArgs(unsigned* curVarNum, unsigned skipArgs, unsigned } #endif - if (info.compIsVarArgs || (opts.compUseSoftFP && varTypeIsFloating(varDsc))) + // Under a soft-float ABI a parameter that lives in a floating-point register + // arrives in an integer one (armel); a parameter whose type is itself in the + // integer register file (RISC-V lp64) is an ordinary integer parameter. + if (info.compIsVarArgs || (opts.compUseSoftFP && varTypeIsFloating(varDsc) && varTypeUsesFloatReg(varDsc))) { #ifndef TARGET_X86 // TODO-CQ: We shouldn't have to go as far as to declare these @@ -897,7 +900,7 @@ void Compiler::lvaInitVarDsc(LclVarDsc* varDsc, } var_types type = JITtype2varType(corInfoType); - if (varTypeIsFloating(type)) + if (varTypeIsFloating(type) && varTypeUsesFloatReg(type)) { compFloatingPointUsed = true; } diff --git a/src/coreclr/jit/lower.cpp b/src/coreclr/jit/lower.cpp index ce5d10adeb9756..649c926f82d018 100644 --- a/src/coreclr/jit/lower.cpp +++ b/src/coreclr/jit/lower.cpp @@ -5478,7 +5478,10 @@ void Lowering::LowerFieldListToFieldListOfRegisters(GenTreeFieldList* fieldLis } // If this is a float -> int insertion, then we need the bitcast now. - if (varTypeUsesFloatReg(value) && varTypeUsesIntReg(regInfo.RegType)) + // (Also checked by type: under a soft-float ABI the FP value already + // lives in an integer register but still needs the integer view for + // the widening and shifting below.) + if ((varTypeUsesFloatReg(value) || varTypeIsFloating(value)) && varTypeUsesIntReg(regInfo.RegType)) { assert((genTypeSize(value) == 4) || (genTypeSize(value) == 8)); var_types castType = genTypeSize(value) == 4 ? TYP_INT : TYP_LONG; diff --git a/src/coreclr/jit/lsra.cpp b/src/coreclr/jit/lsra.cpp index 5803efd8614e12..e024b1d446a36f 100644 --- a/src/coreclr/jit/lsra.cpp +++ b/src/coreclr/jit/lsra.cpp @@ -1009,6 +1009,14 @@ LinearScan::LinearScan(Compiler* theCompiler) availableRegs[static_cast(TYP_##tn)] = ®Fld; #include "typelist.h" #undef DEF_TP +#ifdef TARGET_RISCV64 + if (m_compiler->opts.compUseSoftFP) + { + // No FP register file: float and double values are allocated integer registers. + availableRegs[TYP_FLOAT] = &availableIntRegs; + availableRegs[TYP_DOUBLE] = &availableIntRegs; + } +#endif // TARGET_RISCV64 // Updating lowGprRegs with final value #if defined(TARGET_XARCH) #if defined(TARGET_AMD64) diff --git a/src/coreclr/jit/lsrabuild.cpp b/src/coreclr/jit/lsrabuild.cpp index 099ae3d07221d1..bfb65435071943 100644 --- a/src/coreclr/jit/lsrabuild.cpp +++ b/src/coreclr/jit/lsrabuild.cpp @@ -1105,7 +1105,7 @@ bool LinearScan::buildKillPositionsForNode(GenTree* tree, LsraLocation currentLo } else #endif // FEATURE_PARTIAL_SIMD_CALLEE_SAVE - if (varTypeIsFloating(varDsc) && + if (varTypeIsFloating(varDsc) && varTypeUsesFloatReg(varDsc) && !VarSetOps::IsMember(m_compiler, fpCalleeSaveCandidateVars, varIndex)) { continue; @@ -4330,6 +4330,15 @@ int LinearScan::BuildReturn(GenTree* tree) else #endif // FEATURE_MULTIREG_RET { +#ifdef TARGET_RISCV64 + if (varTypeIsFloating(tree) && !varTypeUsesFloatReg(tree)) + { + // Soft-float: FP values are returned in the integer return register. + BuildUse(op1, RBM_INTRET.GetIntRegSet()); + return 1; + } +#endif // TARGET_RISCV64 + // Non-struct type return - determine useCandidates switch (tree->TypeGet()) { diff --git a/src/coreclr/jit/lsrariscv64.cpp b/src/coreclr/jit/lsrariscv64.cpp index 2531ce69bf4605..656d7288bb2f45 100644 --- a/src/coreclr/jit/lsrariscv64.cpp +++ b/src/coreclr/jit/lsrariscv64.cpp @@ -145,7 +145,9 @@ int LinearScan::BuildNode(GenTree* tree) { emitAttr size = emitActualTypeSize(tree); int64_t bits; - if (emitter::isSingleInstructionFpImm(tree->AsDblCon()->DconValue(), size, &bits) && bits != 0) + // Under soft-float the bit pattern is materialized directly into the (integer) target register. + if (!m_compiler->opts.compUseSoftFP && + emitter::isSingleInstructionFpImm(tree->AsDblCon()->DconValue(), size, &bits) && bits != 0) { buildInternalIntRegisterDefForNode(tree); buildInternalRegisterUses(); @@ -858,7 +860,7 @@ int LinearScan::BuildIndir(GenTreeIndir* indirTree) addr->AsIntCon()->FitsInAddrBase(m_compiler) && addr->AsIntCon()->AddrNeedsReloc(m_compiler); if (needsReloc || !emitter::isValidSimm12(indirTree->Offset())) { - bool needTemp = indirTree->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeIsFloating(indirTree); + bool needTemp = indirTree->OperIs(GT_STOREIND, GT_NULLCHECK) || varTypeUsesFloatReg(indirTree); if (needTemp) { // This offset can't be contained in the ld/sd instruction, so we need an internal register diff --git a/src/coreclr/jit/morph.cpp b/src/coreclr/jit/morph.cpp index 28092b740b13fa..f5d3d34cee3668 100644 --- a/src/coreclr/jit/morph.cpp +++ b/src/coreclr/jit/morph.cpp @@ -253,6 +253,337 @@ GenTree* Compiler::fgMorphIntoHelperCall(GenTree* tree, int helper, bool morphAr return tree; } +#ifdef TARGET_RISCV64 + +//------------------------------------------------------------------------ +// Soft-float (RISC-V without the F/D extensions) +// +// Under opts.compUseSoftFP there are no floating-point instructions or +// registers: float/double values live in integer registers (see +// Compiler::compInitOptions), and every floating-point arithmetic operation, +// comparison and conversion is expanded during global morph into a call to +// one of the soft-float helpers, so that no such node survives into the +// backend. The IR keeps its floating-point types, so value numbering, +// constant folding and CSE work unchanged; the helpers are modelled as the +// operations they implement (see fgValueNumberJitHelperMethodVNFunc). +// + +//------------------------------------------------------------------------ +// fgMorphSoftFloatArith: expand an FP arithmetic operation into a helper call +// +// Arguments: +// tree - the GT_ADD, GT_SUB, GT_MUL or GT_DIV node +// +// Return Value: +// The morphed replacement tree. +// +// Notes: +// Unlike USE_HELPER_FOR_ARITH this builds a new call node: the importer +// only allocates add/sub as small nodes, which cannot be turned into a +// call in place. +// +GenTree* Compiler::fgMorphSoftFloatArith(GenTreeOp* tree) +{ + assert(opts.compUseSoftFP && tree->OperIs(GT_ADD, GT_SUB, GT_MUL, GT_DIV) && varTypeIsFloating(tree)); + + const var_types type = tree->TypeGet(); + + // The IL stack has a single F type, so the operands may differ in size + // from the result; the helpers take operands of the result type. + for (GenTree** use : {&tree->gtOp1, &tree->gtOp2}) + { + if ((*use)->TypeGet() != type) + { + *use = gtNewCastNode(type, *use, false, type); + } + } + + GenTree* folded = gtFoldExpr(tree); + if (folded != tree) + { + return fgMorphTree(folded); + } + if (folded->OperIsLeaf()) + { + return fgMorphLeaf(folded); + } + + const bool isFloat = (type == TYP_FLOAT); + CorInfoHelpFunc helper; + switch (tree->OperGet()) + { + case GT_ADD: + helper = isFloat ? CORINFO_HELP_FLTADD : CORINFO_HELP_DBLADD; + break; + case GT_SUB: + helper = isFloat ? CORINFO_HELP_FLTSUB : CORINFO_HELP_DBLSUB; + break; + case GT_MUL: + helper = isFloat ? CORINFO_HELP_FLTMUL : CORINFO_HELP_DBLMUL; + break; + default: + helper = isFloat ? CORINFO_HELP_FLTDIV : CORINFO_HELP_DBLDIV; + break; + } + + GenTreeCall* call = gtNewHelperCallNode(helper, type, tree->gtGetOp1(), tree->gtGetOp2()); + return fgMorphTree(call); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatCast: expand a conversion into a helper call +// +// Arguments: +// tree - the cast node +// helper - the conversion helper +// oper - the (possibly widened) operand, tree's cast operand +// +// Return Value: +// The morphed replacement tree. +// +// Notes: +// The counterpart of fgMorphCastIntoHelper that builds a new node: casts +// created after import (impImplicitR4orR8Cast, the widening in +// fgMorphExpandCast) are not large enough to become a call in place. +// +GenTree* Compiler::fgMorphSoftFloatCast(GenTreeCast* tree, CorInfoHelpFunc helper, GenTree* oper) +{ + assert(opts.compUseSoftFP && (tree->CastOp() == oper)); + + if (oper->OperIsConst()) + { + GenTree* folded = gtFoldExprConst(tree); + if (folded != tree) + { + return fgMorphTree(folded); + } + if (folded->OperIsConst()) + { + return fgMorphConst(folded); + } + } + + GenTreeCall* call = gtNewHelperCallNode(helper, genActualType(tree->CastToType()), oper); + return fgMorphTree(call); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatNeg: expand an FP negation into a sign-bit flip +// +// Arguments: +// neg - the GT_NEG node +// +// Return Value: +// The replacement tree, not yet morphed. +// +GenTree* Compiler::fgMorphSoftFloatNeg(GenTreeOp* neg) +{ + assert(opts.compUseSoftFP && neg->OperIs(GT_NEG) && varTypeIsFloating(neg)); + + const var_types type = neg->TypeGet(); + GenTree* op = neg->gtGetOp1(); + + if (op->IsCnsFltOrDbl()) + { + return gtNewDconNode(-op->AsDblCon()->DconValue(), type); + } + + // -x == x ^ signBit, computed in the integer domain. + const var_types intType = (type == TYP_FLOAT) ? TYP_INT : TYP_LONG; + GenTree* signBit = (type == TYP_FLOAT) ? gtNewIconNode(INT32_MIN, TYP_INT) : gtNewLconNode(INT64_MIN); + GenTree* bits = gtNewOperNode(GT_XOR, intType, gtNewBitCastNode(intType, op), signBit); + return gtNewBitCastNode(type, bits); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatRelop: expand an FP comparison into a compare helper call +// +// Arguments: +// relop - the comparison node +// +// Return Value: +// The replacement tree, an integer comparison of the helper result, not yet morphed. +// +// Notes: +// The helpers return a three-way result and differ only for unordered +// operands (CMP_LE returns 1, CMP_GE returns -1), which lets each IL form +// be expressed with a single call and a comparison of its result with 0: +// +// oeq: LE == 0 une: LE != 0 +// olt: LE < 0 ult: GE < 0 +// ole: LE <= 0 ule: GE <= 0 +// ogt: GE > 0 ugt: LE > 0 +// oge: GE >= 0 uge: LE >= 0 +// +// ueq and one need both results: +// +// ueq = (LE >= 0) && (GE <= 0) one = (LE < 0) || (GE > 0) +// +GenTree* Compiler::fgMorphSoftFloatRelop(GenTreeOp* relop) +{ + assert(opts.compUseSoftFP && relop->OperIsCompare()); + + GenTree* op1 = relop->gtGetOp1(); + GenTree* op2 = relop->gtGetOp2(); + assert(varTypeIsFloating(op1) && (op1->TypeGet() == op2->TypeGet())); + + const bool isFloat = op1->TypeIs(TYP_FLOAT); + const bool isUnordered = (relop->gtFlags & GTF_RELOP_NAN_UN) != 0; + const genTreeOps oper = relop->OperGet(); + const CorInfoHelpFunc cmpLE = isFloat ? CORINFO_HELP_FLTCMP_LE : CORINFO_HELP_DBLCMP_LE; + const CorInfoHelpFunc cmpGE = isFloat ? CORINFO_HELP_FLTCMP_GE : CORINFO_HELP_DBLCMP_GE; + + GenTree* result; + if (oper == (isUnordered ? GT_EQ : GT_NE)) + { + // ueq / one: both helpers are needed, so the operands go into temps. + // The stores are placed in the first operand of the AND/OR; the two + // operands both contain calls, so their evaluation order is fixed. + TempInfo tmp1 = fgMakeTemp(op1); + TempInfo tmp2 = fgMakeTemp(op2); + + GenTree* le = gtNewHelperCallNode(cmpLE, TYP_INT, tmp1.load, tmp2.load); + le = gtNewOperNode(GT_COMMA, TYP_INT, tmp1.store, gtNewOperNode(GT_COMMA, TYP_INT, tmp2.store, le)); + GenTree* ge = gtNewHelperCallNode(cmpGE, TYP_INT, gtCloneExpr(tmp1.load), gtCloneExpr(tmp2.load)); + + GenTree* cmp; + if (oper == GT_EQ) + { + cmp = gtNewOperNode(GT_AND, TYP_INT, gtNewOperNode(GT_GE, TYP_INT, le, gtNewIconNode(0)), + gtNewOperNode(GT_LE, TYP_INT, ge, gtNewIconNode(0))); + } + else + { + cmp = gtNewOperNode(GT_OR, TYP_INT, gtNewOperNode(GT_LT, TYP_INT, le, gtNewIconNode(0)), + gtNewOperNode(GT_GT, TYP_INT, ge, gtNewIconNode(0))); + } + // Keep the tree rooted at a comparison, for GT_JTRUE users. + result = gtNewOperNode(GT_NE, TYP_INT, cmp, gtNewIconNode(0)); + } + else + { + CorInfoHelpFunc helper; + switch (oper) + { + case GT_EQ: + case GT_NE: + helper = cmpLE; + break; + case GT_LT: + case GT_LE: + helper = isUnordered ? cmpGE : cmpLE; + break; + default: + assert((oper == GT_GT) || (oper == GT_GE)); + helper = isUnordered ? cmpLE : cmpGE; + break; + } + GenTree* call = gtNewHelperCallNode(helper, TYP_INT, op1, op2); + result = gtNewOperNode(oper, TYP_INT, call, gtNewIconNode(0)); + } + + // A JTRUE/QMARK condition carries GTF_RELOP_JMP_USED and GTF_DONT_CSE (set by + // the parent before its operands are morphed); the replacement must keep both + // so that CSE does not turn the condition into a local. + result->gtFlags |= (relop->gtFlags & (GTF_RELOP_JMP_USED | GTF_DONT_CSE)); + return result; +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatCkFinite: expand GT_CKFINITE into an exponent check +// +// Arguments: +// ckFinite - the GT_CKFINITE node +// +// Return Value: +// The replacement tree, not yet morphed: +// +// COMMA(tmp = x, COMMA(BOUNDS_CHECK((bits(tmp) >> expShift) & expMask, expMask), tmp)) +// +// NaN and the infinities have all exponent bits set, so the bounds check +// (index >= length) throws exactly for them. +// +GenTree* Compiler::fgMorphSoftFloatCkFinite(GenTreeOp* ckFinite) +{ + assert(opts.compUseSoftFP && ckFinite->OperIs(GT_CKFINITE)); + + const var_types type = ckFinite->TypeGet(); + const bool isFloat = (type == TYP_FLOAT); + const var_types intType = isFloat ? TYP_INT : TYP_LONG; + const int expBits = isFloat ? 8 : 11; + const int expMask = (1 << expBits) - 1; + + TempInfo tmp = fgMakeTemp(ckFinite->gtGetOp1()); + GenTree* bits = gtNewBitCastNode(intType, tmp.load); + GenTree* exp = gtNewOperNode(GT_RSZ, intType, bits, gtNewIconNode(genTypeSize(type) * 8 - 1 - expBits)); + exp = gtNewOperNode(GT_AND, intType, exp, isFloat ? gtNewIconNode(expMask) : gtNewLconNode(expMask)); + if (!isFloat) + { + exp = gtNewCastNode(TYP_INT, exp, false, TYP_INT); + } + GenTree* check = new (this, GT_BOUNDS_CHECK) GenTreeBoundsChk(exp, gtNewIconNode(expMask), SCK_ARITH_EXCPN); + + GenTree* result = gtNewOperNode(GT_COMMA, type, check, gtCloneExpr(tmp.load)); + return gtNewOperNode(GT_COMMA, type, tmp.store, result); +} + +//------------------------------------------------------------------------ +// fgMorphSoftFloatCastToInt32: expand a non-overflow double -> int/uint cast +// +// Arguments: +// src - the TYP_DOUBLE source +// toUnsigned - true for uint +// +// Return Value: +// The replacement tree, not yet morphed. +// +// Notes: +// The 64-bit conversion helpers implement the .NET semantics (NaN -> 0, +// saturation); their result is saturated to the 32-bit range here. The +// clamping is branchless and does not use GT_SELECT, which on RISC-V needs +// Zicond. The temps are stored first in an explicit COMMA chain since the +// store-before-use dependency is not otherwise expressed in the IR. +// +GenTree* Compiler::fgMorphSoftFloatCastToInt32(GenTree* src, bool toUnsigned) +{ + assert(opts.compUseSoftFP && src->TypeIs(TYP_DOUBLE)); + + GenTreeCall* cvt = gtNewHelperCallNode(toUnsigned ? CORINFO_HELP_DBL2ULNG : CORINFO_HELP_DBL2LNG, TYP_LONG, src); + TempInfo tmpVal = fgMakeTemp(cvt); + + if (toUnsigned) + { + // result = (uint)val | -(val > UINT32_MAX); the helper never returns a negative value. + GenTree* isHi = gtNewOperNode(GT_GT, TYP_INT, gtCloneExpr(tmpVal.load), gtNewLconNode((int64_t)UINT32_MAX)); + isHi->gtFlags |= GTF_UNSIGNED; + GenTree* trunc = gtNewCastNode(TYP_INT, tmpVal.load, false, TYP_UINT); + GenTree* result = gtNewOperNode(GT_OR, TYP_INT, trunc, gtNewOperNode(GT_NEG, TYP_INT, isHi)); + return gtNewOperNode(GT_COMMA, TYP_INT, tmpVal.store, result); + } + + // maskHi = -(val > INT32_MAX); maskLo = -(val < INT32_MIN); + // result = ((int)val & ~(maskHi | maskLo)) | (INT32_MAX & maskHi) | (INT32_MIN & maskLo) + TempInfo tmpHi = + fgMakeTemp(gtNewOperNode(GT_NEG, TYP_INT, + gtNewOperNode(GT_GT, TYP_INT, gtCloneExpr(tmpVal.load), gtNewLconNode(INT32_MAX)))); + TempInfo tmpLo = + fgMakeTemp(gtNewOperNode(GT_NEG, TYP_INT, + gtNewOperNode(GT_LT, TYP_INT, gtCloneExpr(tmpVal.load), gtNewLconNode(INT32_MIN)))); + + GenTree* trunc = gtNewCastNode(TYP_INT, tmpVal.load, false, TYP_INT); + GenTree* outMask = gtNewOperNode(GT_OR, TYP_INT, tmpHi.load, tmpLo.load); + GenTree* inVal = gtNewOperNode(GT_AND, TYP_INT, trunc, gtNewOperNode(GT_NOT, TYP_INT, outMask)); + GenTree* hiVal = gtNewOperNode(GT_AND, TYP_INT, gtNewIconNode(INT32_MAX, TYP_INT), gtCloneExpr(tmpHi.load)); + GenTree* loVal = gtNewOperNode(GT_AND, TYP_INT, gtNewIconNode(INT32_MIN, TYP_INT), gtCloneExpr(tmpLo.load)); + GenTree* result = gtNewOperNode(GT_OR, TYP_INT, gtNewOperNode(GT_OR, TYP_INT, inVal, hiVal), loVal); + + result = gtNewOperNode(GT_COMMA, TYP_INT, tmpLo.store, result); + result = gtNewOperNode(GT_COMMA, TYP_INT, tmpHi.store, result); + return gtNewOperNode(GT_COMMA, TYP_INT, tmpVal.store, result); +} + +#endif // TARGET_RISCV64 + //------------------------------------------------------------------------ // fgMorphExpandCast: Performs the pre-order (required) morphing for a cast. // @@ -282,6 +613,7 @@ GenTree* Compiler::fgMorphIntoHelperCall(GenTree* tree, int helper, bool morphAr // in which case the cast may be transformed into an unchecked one // and its operand changed (the cast "expanded" into two). // + GenTree* Compiler::fgMorphExpandCast(GenTreeCast* tree) { GenTree* oper = tree->CastOp(); @@ -445,6 +777,23 @@ GenTree* Compiler::fgMorphExpandCast(GenTreeCast* tree) case TYP_ULONG: helper = CORINFO_HELP_DBL2ULNG; break; +#ifdef TARGET_RISCV64 + case TYP_INT: + case TYP_UINT: + { + // Soft-float: convert with the 64-bit helper and saturate to 32 bits. + assert(opts.compUseSoftFP); + if (tree->CastOp()->OperIsConst()) + { + GenTree* folded = gtFoldExprConst(tree); + if (folded != tree) + { + return fgMorphTree(folded); + } + } + return fgMorphTree(fgMorphSoftFloatCastToInt32(oper, dstType == TYP_UINT)); + } +#endif // TARGET_RISCV64 default: unreached(); } @@ -469,6 +818,42 @@ GenTree* Compiler::fgMorphExpandCast(GenTreeCast* tree) return fgMorphTree(oper); } +#ifdef TARGET_RISCV64 + else if (opts.compUseSoftFP && varTypeIsFloating(dstType)) + { + // Soft-float: conversions to floating point are helper calls. + if (varTypeIsFloating(srcType)) + { + if (srcType == dstType) + { + return fgMorphTree(oper); + } + return fgMorphSoftFloatCast(tree, (dstType == TYP_DOUBLE) ? CORINFO_HELP_FLT2DBL : CORINFO_HELP_DBL2FLT, + oper); + } + + assert(varTypeIsIntegral(srcType)); + if (!varTypeIsLong(srcType)) + { + // Widen to 64 bits first; a zero-extended uint is exact in the signed helper. + oper = gtNewCastNode(TYP_LONG, oper, tree->IsUnsigned(), TYP_LONG); + tree->ClearUnsigned(); + tree->CastOp() = oper; + } + + CorInfoHelpFunc helper; + if (dstType == TYP_FLOAT) + { + helper = tree->IsUnsigned() ? CORINFO_HELP_ULNG2FLT : CORINFO_HELP_LNG2FLT; + } + else + { + helper = tree->IsUnsigned() ? CORINFO_HELP_ULNG2DBL : CORINFO_HELP_LNG2DBL; + } + return fgMorphSoftFloatCast(tree, helper, oper); + } +#endif // TARGET_RISCV64 + #ifndef TARGET_64BIT else if (varTypeIsLong(srcType)) { @@ -7011,6 +7396,40 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) // Some arithmetic operators need to use a helper call to the EE int helper; +#ifdef TARGET_RISCV64 + // Soft-float: expand the FP operations into helper calls before morphing their operands. + case GT_ADD: + case GT_SUB: + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphSoftFloatArith(tree->AsOp()); + } + break; + + case GT_NEG: + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphTree(fgMorphSoftFloatNeg(tree->AsOp())); + } + break; + + case GT_LT: + case GT_LE: + case GT_GE: + if (opts.compUseSoftFP && varTypeIsFloating(op1)) + { + return fgMorphTree(fgMorphSoftFloatRelop(tree->AsOp())); + } + break; + + case GT_CKFINITE: + if (opts.compUseSoftFP) + { + return fgMorphTree(fgMorphSoftFloatCkFinite(tree->AsOp())); + } + break; +#endif // TARGET_RISCV64 + case GT_STORE_LCL_VAR: case GT_STORE_LCL_FLD: { @@ -7085,6 +7504,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) case GT_MUL: noway_assert(op2 != nullptr); +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphSoftFloatArith(tree->AsOp()); + } +#endif // TARGET_RISCV64 + #if !defined(TARGET_64BIT) && !defined(TARGET_WASM) if (typ == TYP_LONG) { @@ -7189,6 +7615,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) break; case GT_DIV: +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(typ)) + { + return fgMorphSoftFloatArith(tree->AsOp()); + } +#endif // TARGET_RISCV64 + // Convert DIV to UDIV if both op1 and op2 are known to be never negative if (varTypeIsIntegral(tree) && op1->IsNeverNegative(this) && op2->IsNeverNegative(this)) { @@ -7484,6 +7917,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) case GT_EQ: case GT_NE: { +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(op1)) + { + return fgMorphTree(fgMorphSoftFloatRelop(tree->AsOp())); + } +#endif // TARGET_RISCV64 + if (opts.OptimizationEnabled()) { GenTree* optimizedTree = gtFoldTypeCompare(tree); @@ -7545,6 +7985,13 @@ GenTree* Compiler::fgMorphSmpOp(GenTree* tree, bool* optAssertionPropDone) case GT_GT: { +#ifdef TARGET_RISCV64 + if (opts.compUseSoftFP && varTypeIsFloating(op1)) + { + return fgMorphTree(fgMorphSoftFloatRelop(tree->AsOp())); + } +#endif // TARGET_RISCV64 + // Try and optimize nullable boxes feeding compares GenTree* optimizedTree = gtFoldBoxNullable(tree); diff --git a/src/coreclr/jit/utils.cpp b/src/coreclr/jit/utils.cpp index 5e835a26bc190f..f2724acf343dce 100644 --- a/src/coreclr/jit/utils.cpp +++ b/src/coreclr/jit/utils.cpp @@ -86,7 +86,11 @@ const BYTE varTypeClassification[] = { #undef DEF_TP }; +#ifdef TARGET_RISCV64 +BYTE varTypeRegister[] = { +#else const BYTE varTypeRegister[] = { +#endif #define DEF_TP(tn, nm, jitType, sz, sze, asze, st, al, regTyp, regFld, csr, ctr, tf) regTyp, #include "typelist.h" #undef DEF_TP diff --git a/src/coreclr/jit/vartype.h b/src/coreclr/jit/vartype.h index c501dea656ebda..e89f7e5518e8bd 100644 --- a/src/coreclr/jit/vartype.h +++ b/src/coreclr/jit/vartype.h @@ -48,7 +48,13 @@ enum var_types_register /*****************************************************************************/ const extern BYTE varTypeClassification[TYP_COUNT]; +#ifdef TARGET_RISCV64 +// Not const: a soft-float (no F/D) process moves TYP_FLOAT/TYP_DOUBLE to the +// integer register file, see Compiler::compInitOptions. +extern BYTE varTypeRegister[TYP_COUNT]; +#else const extern BYTE varTypeRegister[TYP_COUNT]; +#endif // make any class with a TypeGet member also have a function TypeGet() that does the same thing template @@ -339,6 +345,10 @@ inline bool varTypeUsesFloatArgReg(T vt) // Exception: Windows arm64 native varargs passes them using general-purpose (integer) registers or // by value on the stack, or split between registers and stack. return varTypeUsesFloatReg(vt); +#elif defined(TARGET_RISCV64) + // The register class of TYP_FLOAT/TYP_DOUBLE follows the ABI: under the soft-float + // (lp64) ABI they live in, and are returned in, integer registers. + return varTypeUsesFloatReg(vt); #else // Other targets pass them as regular structs - by reference or by value. return varTypeIsFloating(vt); diff --git a/src/coreclr/nativeaot/Runtime/MathHelpers.cpp b/src/coreclr/nativeaot/Runtime/MathHelpers.cpp index 2cd1075b047502..f33d13ada2031b 100644 --- a/src/coreclr/nativeaot/Runtime/MathHelpers.cpp +++ b/src/coreclr/nativeaot/Runtime/MathHelpers.cpp @@ -9,26 +9,22 @@ // Floating point and 64-bit integer math helpers. // +// The .NET semantics of the floating point to integer conversions (NaN -> 0, +// saturation) are spelled out rather than left to the C cast: the cast only +// happens to have them where the hardware instruction does, and a soft-float +// build lowers it to a compiler-rt routine that has neither. FCIMPL1_D(uint64_t, RhpDbl2ULng, double val) { -#if defined(HOST_X86) || defined(HOST_AMD64) const double uint64_max_plus_1 = 4294967296.0 * 4294967296.0; return (val > 0) ? ((val >= uint64_max_plus_1) ? UINT64_MAX : (uint64_t)val) : 0; -#else - return (uint64_t)val; -#endif } FCIMPLEND FCIMPL1_D(int64_t, RhpDbl2Lng, double val) { -#if defined(HOST_X86) || defined(HOST_AMD64) || defined(HOST_ARM) const double int64_min = -2147483648.0 * 4294967296.0; const double int64_max = 2147483648.0 * 4294967296.0; return (val != val) ? 0 : (val <= int64_min) ? INT64_MIN : (val >= int64_max) ? INT64_MAX : (int64_t)val; -#else - return (int64_t)val; -#endif } FCIMPLEND @@ -61,6 +57,12 @@ FCIMPL2_LL(uint64_t, ModUInt64Internal, uint64_t i, uint64_t j) } FCIMPLEND +#endif + +// The int64 -> floating point conversion helpers are used where the JIT cannot +// emit the conversion inline: 32-bit targets and the soft-float RISC-V ABI. +#if !defined(HOST_64BIT) || defined(HOST_RISCV64) + FCIMPL1_L(double, RhpLng2Dbl, int64_t val) { return (double)val; From c42bdf341d5f80ac9a34d9098b07d6d011184935 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 9 Sep 2026 11:27:26 +0100 Subject: [PATCH 11/16] [RISC-V] Add a floating-point semantics test for the soft-float target Bit-exact expectations for arithmetic rounding, NaN comparison semantics including the unordered branch forms, saturating and checked conversions for every integer width, conversions from every integer width, float <-> double, negation, remainder, the Math intrinsics that become calls, CSE of repeated helper calls, values across call boundaries including stack-passed arguments, and float-bits-as-int sign extension. Every expected value is the exact IEEE 754 result, so the test is valid on every target and passes on hard-float today; on a soft-float image the same assertions exercise the compiler-rt helpers. Signed-off-by: Maxim Menshikov --- src/tests/JIT/Directed/softfloat/SoftFloat.cs | 284 ++++++++++++++++++ .../JIT/Directed/softfloat/SoftFloat.csproj | 12 + 2 files changed, 296 insertions(+) create mode 100644 src/tests/JIT/Directed/softfloat/SoftFloat.cs create mode 100644 src/tests/JIT/Directed/softfloat/SoftFloat.csproj diff --git a/src/tests/JIT/Directed/softfloat/SoftFloat.cs b/src/tests/JIT/Directed/softfloat/SoftFloat.cs new file mode 100644 index 00000000000000..e6f7f32d87c918 --- /dev/null +++ b/src/tests/JIT/Directed/softfloat/SoftFloat.cs @@ -0,0 +1,284 @@ +// Licensed to the .NET Foundation under one or more agreements. +// The .NET Foundation licenses this file to you under the MIT license. +// +// Floating-point semantics that a soft-float target (RISC-V lp64, where the +// JIT expands every FP operation into a helper call) must reproduce bit for +// bit. All expected values are exact IEEE 754 results, so the test is equally +// valid on hard-float targets: arithmetic rounding, comparison semantics with +// NaN (including the unordered branch forms), the saturating .NET conversions +// to every integer width, the checked conversions, conversions from every +// integer width, float <-> double, negation, remainder, Math intrinsics that +// become calls, values passed and returned across call boundaries in integer +// registers, and the sign extension of float bits reinterpreted as int. + +using System; +using System.Runtime.CompilerServices; +using Xunit; + +public class SoftFloat +{ + static int s_failures; + + static void Check(bool ok, string name) + { + if (!ok) + { + s_failures++; + Console.WriteLine("FAIL: " + name); + } + } + + static void CheckBits(double actual, ulong expected, string name) + => Check((ulong)BitConverter.DoubleToInt64Bits(actual) == expected, + name + ": " + BitConverter.DoubleToInt64Bits(actual).ToString("X16") + " != " + expected.ToString("X16")); + + static void CheckBits(float actual, uint expected, string name) + => Check((uint)BitConverter.SingleToInt32Bits(actual) == expected, + name + ": " + BitConverter.SingleToInt32Bits(actual).ToString("X8") + " != " + expected.ToString("X8")); + + [MethodImpl(MethodImplOptions.NoInlining)] static double D(double x) => x; + [MethodImpl(MethodImplOptions.NoInlining)] static float F(float x) => x; + [MethodImpl(MethodImplOptions.NoInlining)] static int I(int x) => x; + [MethodImpl(MethodImplOptions.NoInlining)] static long L(long x) => x; + + static void Arithmetic() + { + double a = D(0.1), b = D(0.2); + CheckBits(a + b, 0x3FD3333333333334, "0.1 + 0.2"); + CheckBits(D(1.0) / D(3.0), 0x3FD5555555555555, "1 / 3"); + CheckBits(D(100.0) / D(7.0), 0x402C924924924925, "100 / 7"); + CheckBits(D(3.7) * D(2.1), 0x401F147AE147AE15, "3.7 * 2.1"); + CheckBits(D(1e308) * D(10.0), 0x7FF0000000000000, "overflow to +inf"); + CheckBits(D(-1e308) * D(10.0), 0xFFF0000000000000, "overflow to -inf"); + CheckBits(D(double.Epsilon) * D(0.5), 0x0000000000000000, "underflow to 0"); + CheckBits(D(0.0) + D(-0.0), 0x0000000000000000, "0 + -0 = +0"); + CheckBits(D(-0.0) - D(0.0), 0x8000000000000000, "-0 - 0 = -0"); + CheckBits(D(-0.0) * D(1.0), 0x8000000000000000, "-0 * 1 = -0"); + Check(double.IsNaN(D(double.PositiveInfinity) - D(double.PositiveInfinity)), "inf - inf"); + Check(double.IsNaN(D(0.0) / D(0.0)), "0 / 0"); + CheckBits(D(1.0) / D(0.0), 0x7FF0000000000000, "1 / 0"); + CheckBits(D(-1.0) / D(0.0), 0xFFF0000000000000, "-1 / 0"); + CheckBits(D(9007199254740992.0) + D(1.0), 0x4340000000000000, "2^53 + 1 rounds to even"); + + float fa = F(0.1f), fb = F(0.2f); + CheckBits(fa + fb, 0x3E99999Au, "0.1f + 0.2f"); + CheckBits(F(1.0f) / F(3.0f), 0x3EAAAAABu, "1f / 3f"); + CheckBits(F(7.0f) / F(3.0f), 0x40155555u, "7f / 3f"); + CheckBits(F(1e10f) * F(1e10f), 0x60AD78ECu, "1e10f * 1e10f"); + CheckBits(F(float.MaxValue) * F(2.0f), 0x7F800000u, "float overflow to +inf"); + CheckBits(F(-0.0f) * F(1.0f), 0x80000000u, "-0f * 1f = -0f"); + Check(float.IsNaN(F(0.0f) / F(0.0f)), "0f / 0f"); + + // Remainder goes through the pre-existing helpers. + Check(D(5.5) % D(2.0) == 1.5, "5.5 % 2"); + Check(D(-5.5) % D(2.0) == -1.5, "-5.5 % 2"); + Check(D(5.5) % D(double.PositiveInfinity) == 5.5, "x % inf"); + Check(double.IsNaN(D(5.5) % D(0.0)), "x % 0"); + Check(F(5.5f) % F(2.0f) == 1.5f, "5.5f % 2f"); + + // The same expression twice: CSE / value numbering of the helper calls. + double c1 = a * b + a / b; + double c2 = a * b + a / b; + Check(BitConverter.DoubleToInt64Bits(c1) == BitConverter.DoubleToInt64Bits(c2), "CSE"); + } + + static void Negation() + { + CheckBits(-D(1.5), 0xBFF8000000000000, "-1.5"); + CheckBits(-D(0.0), 0x8000000000000000, "-(+0)"); + CheckBits(-D(-0.0), 0x0000000000000000, "-(-0)"); + Check(double.IsNaN(-D(double.NaN)), "-NaN"); + CheckBits(-D(double.PositiveInfinity), 0xFFF0000000000000, "-inf"); + CheckBits(-F(1.5f), 0xBFC00000u, "-1.5f"); + CheckBits(-F(0.0f), 0x80000000u, "-(+0f)"); + Check(float.IsNaN(-F(float.NaN)), "-NaNf"); + } + + static void Comparisons() + { + double nan = D(double.NaN), one = D(1.0), two = D(2.0), nz = D(-0.0), pz = D(0.0); + + Check(!(nan == nan), "NaN == NaN"); + Check(nan != nan, "NaN != NaN"); + Check(!(nan < one) && !(nan <= one) && !(nan > one) && !(nan >= one), "NaN ordered compares"); + Check(!(one < nan) && !(one <= nan) && !(one > nan) && !(one >= nan), "ordered compares with NaN rhs"); + Check(one < two && one <= two && two > one && two >= one && one <= one && one >= one, "ordered"); + Check(!(two < one) && !(two <= one) && !(one > two) && !(one >= two), "ordered false"); + Check(nz == pz && !(nz < pz) && nz <= pz && nz >= pz, "-0 == +0"); + Check(D(double.NegativeInfinity) < D(double.MinValue) && D(double.MaxValue) < D(double.PositiveInfinity), "infinities"); + + // Unordered branch forms (bge.un etc.) come from the negated conditions. + int hits = 0; + if (!(nan < one)) hits |= 1; // uge + if (!(nan <= one)) hits |= 2; // ugt + if (!(nan > one)) hits |= 4; // ule + if (!(nan >= one)) hits |= 8; // ult + if (!(nan == one)) hits |= 16; // une + Check(hits == 31, "unordered branches with NaN"); + hits = 0; + if (!(one < two)) hits |= 1; + if (!(two <= one)) hits |= 2; + if (!(one > two)) hits |= 4; + if (!(two >= one)) hits |= 8; + if (!(one == two)) hits |= 16; + Check(hits == 2 + 4 + 16, "unordered branches, ordered operands"); + + float fn = F(float.NaN), f1 = F(1.0f), f2 = F(2.0f); + Check(!(fn == fn) && fn != fn && !(fn < f1) && !(fn >= f1), "float NaN compares"); + Check(f1 < f2 && f2 >= f1 && !(f2 <= f1), "float ordered"); + Check(F(-0.0f) == F(0.0f), "-0f == +0f"); + } + + static void ToInteger() + { + // .NET semantics: NaN -> 0, saturation to the range of the destination. + Check((int)D(double.NaN) == 0, "(int)NaN"); + Check((int)D(1e10) == int.MaxValue, "(int)1e10"); + Check((int)D(-1e10) == int.MinValue, "(int)-1e10"); + Check((int)D(2147483647.9) == int.MaxValue, "(int)2147483647.9"); + Check((int)D(-2147483648.9) == int.MinValue, "(int)-2147483648.9"); + Check((int)D(-1.9) == -1 && (int)D(1.9) == 1, "(int) truncation"); + Check((int)D(double.PositiveInfinity) == int.MaxValue && (int)D(double.NegativeInfinity) == int.MinValue, "(int)inf"); + Check((uint)D(-1.0) == 0 && (uint)D(double.NaN) == 0, "(uint) negative/NaN"); + Check((uint)D(5e9) == uint.MaxValue && (uint)D(4294967295.9) == uint.MaxValue, "(uint) saturation"); + Check((uint)D(3.99) == 3 && (uint)D(4294967295.0) == uint.MaxValue, "(uint) values"); + Check((long)D(double.NaN) == 0 && (long)D(1e30) == long.MaxValue && (long)D(-1e30) == long.MinValue, "(long)"); + Check((long)D(-9223372036854775808.0) == long.MinValue && (long)D(9223372036854775807.0) == long.MaxValue, "(long) edges"); + Check((ulong)D(-1.0) == 0 && (ulong)D(1e30) == ulong.MaxValue && (ulong)D(18446744073709551615.0) == ulong.MaxValue, "(ulong)"); + Check((ulong)D(9223372036854775808.0) == 9223372036854775808UL, "(ulong)2^63"); + // Conversions to the small integer types saturate as well (.NET 11). + Check((byte)D(300.0) == 255 && (byte)D(-5.0) == 0 && (byte)D(double.NaN) == 0, "(byte)"); + Check((sbyte)D(-200.0) == -128 && (sbyte)D(200.0) == 127, "(sbyte)"); + Check((short)D(70000.0) == 32767 && (ushort)D(70000.0) == 65535 && (ushort)D(-1.0) == 0, "(short)/(ushort)"); + + Check((int)F(1e10f) == int.MaxValue && (int)F(float.NaN) == 0 && (int)F(-2.5f) == -2, "(int)float"); + Check((long)F(1e30f) == long.MaxValue && (ulong)F(-1.0f) == 0, "(long)/(ulong) float"); + Check((byte)F(300.0f) == 255, "(byte)float"); + + // Checked conversions throw for NaN and out-of-range values. + Check(Throws(() => checked((int)D(1e10))), "checked (int)1e10"); + Check(Throws(() => checked((int)D(double.NaN))), "checked (int)NaN"); + Check(Throws(() => checked((uint)D(-1.0))), "checked (uint)-1"); + Check(Throws(() => checked((long)D(1e30))), "checked (long)1e30"); + Check(Throws(() => checked((byte)D(256.0))), "checked (byte)256"); + Check(checked((int)D(-2147483648.0)) == int.MinValue && checked((int)D(2147483647.0)) == int.MaxValue, "checked (int) edges"); + Check(checked((long)F(1e18f)) == 999999984306749440L, "checked (long)1e18f"); + } + + static void FromInteger() + { + CheckBits((double)I(int.MinValue), 0xC1E0000000000000, "(double)int.MinValue"); + CheckBits((double)(uint)I(-1), 0x41EFFFFFFFE00000, "(double)uint.MaxValue"); + CheckBits((double)L(long.MaxValue), 0x43E0000000000000, "(double)long.MaxValue"); + CheckBits((double)(ulong)L(-1), 0x43F0000000000000, "(double)ulong.MaxValue"); + CheckBits((double)L(long.MinValue), 0xC3E0000000000000, "(double)long.MinValue"); + CheckBits((double)L(9007199254740993), 0x4340000000000000, "(double)(2^53+1) rounds to even"); + CheckBits((float)I(16777217), 0x4B800000u, "(float)16777217"); + CheckBits((float)L(long.MaxValue), 0x5F000000u, "(float)long.MaxValue"); + CheckBits((float)(ulong)L(-1), 0x5F800000u, "(float)ulong.MaxValue"); + CheckBits((float)(uint)I(-1), 0x4F800000u, "(float)uint.MaxValue"); + CheckBits((double)(short)I(-3), 0xC008000000000000, "(double)short"); + CheckBits((double)(byte)I(255), 0x406FE00000000000, "(double)byte"); + } + + static void FloatDouble() + { + CheckBits((double)F(0.1f), 0x3FB99999A0000000, "(double)0.1f"); + CheckBits((float)D(0.1), 0x3DCCCCCDu, "(float)0.1"); + CheckBits((float)D(1e300), 0x7F800000u, "(float)1e300"); + CheckBits((float)D(-1e300), 0xFF800000u, "(float)-1e300"); + CheckBits((float)D(1e-50), 0x00000000u, "(float)1e-50"); + CheckBits((float)D(-0.0), 0x80000000u, "(float)-0.0"); + Check(float.IsNaN((float)D(double.NaN)), "(float)NaN"); + Check(double.IsNaN((double)F(float.NaN)), "(double)NaNf"); + CheckBits((double)(float)D(16777217.0), 0x4170000000000000, "(double)(float)16777217.0"); + } + + static void Bits() + { + // A float in an integer register may carry undefined upper bits; the + // reinterpretation as int must be a proper sign-extended int. + int bits = BitConverter.SingleToInt32Bits(F(-1.5f)); + Check(bits == -1077936128, "SingleToInt32Bits(-1.5f)"); + Check(bits < 0, "float sign bit as int sign"); + Check((long)bits == -1077936128L, "float bits widened"); + Check(BitConverter.Int32BitsToSingle(bits) == -1.5f, "round trip"); + Check(BitConverter.DoubleToInt64Bits(D(-2.0)) == unchecked((long)0xC000000000000000), "DoubleToInt64Bits"); + Check(BitConverter.Int64BitsToDouble(0x3FF0000000000000) == 1.0, "Int64BitsToDouble"); + } + + struct Pair { public double A; public float B; public int C; } + + [MethodImpl(MethodImplOptions.NoInlining)] + static double Sum10(double a, double b, double c, double d, double e, double f, double g, double h, double i, double j) + => a + b + c + d + e + f + g + h + i + j; + + [MethodImpl(MethodImplOptions.NoInlining)] + static float Mixed(int a, float b, long c, double d, float e) => (float)(a + b + c + d + e); + + [MethodImpl(MethodImplOptions.NoInlining)] + static float SumF11(float a, float b, float c, float d, float e, float f, float g, float h, float i, float j, float k) + => a + b + c + d + e + f + g + h + i + j + k; + + [MethodImpl(MethodImplOptions.NoInlining)] + static double Mixed12(float a, double b, int c, float d, double e, long f, float g, double h, float i, double j, float k, double l) + => a + b + c + d + e + f + g + h + i + j + k + l; + + [MethodImpl(MethodImplOptions.NoInlining)] + static Pair MakePair(double a, float b, int c) => new Pair { A = a * 2, B = b * 2, C = c * 2 }; + + static void Calls() + { + Check(Sum10(1, 2, 3, 4, 5, 6, 7, 8, 9, 10) == 55.0, "10 double args (registers and stack)"); + Check(Mixed(1, 2.5f, 3, 4.25, 5.5f) == 16.25f, "mixed int/float args"); + Check(SumF11(1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11) == 66.0f, "11 float args (registers and stack)"); + Check(Mixed12(0.5f, 1.5, 2, 3.5f, 4.5, 5, 6.5f, 7.5, 8.5f, 9.5, 10.5f, 11.5) == 71.0, "12 mixed args on registers and stack"); + Pair p = MakePair(1.5, 2.5f, 3); + Check(p.A == 3.0 && p.B == 5.0f && p.C == 6, "struct with FP fields"); + double[] arr = { 1.5, 2.5, 3.5 }; + double s = 0; + foreach (double v in arr) s += v; + Check(s == 7.5, "array of doubles"); + float[] farr = { 1.5f, -2.5f }; + Check(farr[0] + farr[1] == -1.0f, "array of floats"); + } + + static void Intrinsics() + { + CheckBits(Math.Sqrt(D(2.0)), 0x3FF6A09E667F3BCD, "Sqrt(2)"); + Check(double.IsNaN(Math.Sqrt(D(-1.0))), "Sqrt(-1)"); + CheckBits(Math.Abs(D(-0.0)), 0x0000000000000000, "Abs(-0)"); + Check(Math.Abs(D(-3.5)) == 3.5 && MathF.Abs(F(-3.5f)) == 3.5f, "Abs"); + Check(double.IsNaN(Math.Max(D(double.NaN), D(1.0))) && double.IsNaN(Math.Min(D(1.0), D(double.NaN))), "Max/Min NaN"); + Check(Math.Max(D(1.0), D(2.0)) == 2.0 && Math.Min(D(1.0), D(2.0)) == 1.0, "Max/Min"); + Check(Math.Max(I(3), I(4)) == 4 && Math.Min(L(-1), L(1)) == -1, "integer Max/Min unaffected"); + Check(Math.Floor(D(-1.5)) == -2.0 && Math.Ceiling(D(1.2)) == 2.0 && Math.Round(D(2.5)) == 2.0, "Floor/Ceiling/Round"); + Check(Math.Truncate(D(-1.7)) == -1.0, "Truncate"); + } + + static bool Throws(Func f) + { + try { _ = f(); } catch (OverflowException) { return true; } + return false; + } + + [Fact] + public static int TestEntryPoint() + { + Arithmetic(); + Negation(); + Comparisons(); + ToInteger(); + FromInteger(); + FloatDouble(); + Bits(); + Calls(); + Intrinsics(); + if (s_failures != 0) + { + Console.WriteLine(s_failures + " failure(s)"); + return 1; + } + return 100; + } +} diff --git a/src/tests/JIT/Directed/softfloat/SoftFloat.csproj b/src/tests/JIT/Directed/softfloat/SoftFloat.csproj new file mode 100644 index 00000000000000..c0a7d8b5cf2b6e --- /dev/null +++ b/src/tests/JIT/Directed/softfloat/SoftFloat.csproj @@ -0,0 +1,12 @@ + + + 1 + + + PdbOnly + True + + + + + From fed286d21e99feaabb4750f610c43cdf39dc2c85 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 16 Sep 2026 15:26:33 +0100 Subject: [PATCH 12/16] [RISC-V] Let a build select the lp64 ABI for NativeAOT publishing The ILCompiler is published as a NativeAOT application for the target, and it gets --targetarch from _targetArchitectureWithAbi. On a riscv64 target that value is plain "riscv64", so the object writer stamps EF_RISCV_FLOAT_ABI_DOUBLE; against a soft-float sysroot the link then fails: ld.lld: error: ilc.o: cannot link object files with different floating-point ABI from Scrt1.o Add the riscv64 case next to the armel one that is already there. It is conditional on a property rather than derived from the RID, because linux-musl-riscv64 covers both lp64d and lp64 userspaces - unlike linux-bionic on ARM, where armel follows from the RID alone. Signed-off-by: Maxim Menshikov --- .../Microsoft.DotNet.ILCompiler.SingleEntry.targets | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets b/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets index 368ab3b7b50c3d..be8043b8ee3089 100644 --- a/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets +++ b/src/coreclr/nativeaot/BuildIntegration/Microsoft.DotNet.ILCompiler.SingleEntry.targets @@ -44,6 +44,12 @@ <_targetArchitectureWithAbi>$(_targetArchitecture) <_targetArchitectureWithAbi Condition="'$(_linuxLibcFlavor)' == 'bionic' and '$(_targetArchitecture)' == 'arm'">armel + + + <_targetArchitectureWithAbi Condition="'$(IlcRiscV64SoftFloat)' == 'true' and '$(_targetArchitecture)' == 'riscv64'">riscv64-lp64 From 6b75f502fa1068035c18320ab373d4f312e47276 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 16 Sep 2026 15:26:34 +0100 Subject: [PATCH 13/16] [RISC-V] Pass -mabi=lp64 to the native linker and the shim compiler for riscv64-lp64 The aggregate-executable shim is compiled from C and linked by clang straight from MSBuild; nothing there knows the target ABI, so on a soft-float sysroot the shim object comes out lp64d and lld refuses to link it against Scrt1.o: ld.lld: error: /tmp/dotnet-dev-certs-b44bf8.o: cannot link object files with different floating-point ABI from /crossrootfs/riscv64/usr/lib/Scrt1.o Derive the flag from _targetArchitectureWithAbi, the same property that already selects --targetarch for ilc, and mark it Shim="true" so both the shim and the main link get it. Signed-off-by: Maxim Menshikov --- .../BuildIntegration/Microsoft.NETCore.Native.Unix.targets | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets b/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets index 5b6862703de826..fb685cced1409a 100644 --- a/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets +++ b/src/coreclr/nativeaot/BuildIntegration/Microsoft.NETCore.Native.Unix.targets @@ -261,6 +261,12 @@ The .NET Foundation licenses this file to you under the MIT license. + + From 2fdd7d6f246875cebe44a82a2ba90de1cb6bedb4 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 16 Sep 2026 16:10:39 +0100 Subject: [PATCH 14/16] [RISC-V] Lower Interlocked and CmpXchg without the A extension genLockedInstructions and genCodeForCmpXchg emit amo* and lr/sc unconditionally. A build that selects an ISA without the A extension gets instructions the target cannot execute, on any Interlocked call. Gate them on InstructionSet_A. Without it, lower to a plain read/modify/write and, for CmpXchg, to a compare and store with no reservation pair. This is the only lowering the ISA allows, and it is correct on a target with a single hart and no preemption. BuildNode in lsrariscv64 extends the address and data lifetimes for the plain sequence, which reuses both registers after the first instruction, and gives the arithmetic and bitwise forms one scratch register for the new value. Signed-off-by: Maxim Menshikov --- src/coreclr/jit/codegenriscv64.cpp | 79 ++++++++++++++++++++++++++---- src/coreclr/jit/lsrariscv64.cpp | 28 +++++++++-- 2 files changed, 95 insertions(+), 12 deletions(-) diff --git a/src/coreclr/jit/codegenriscv64.cpp b/src/coreclr/jit/codegenriscv64.cpp index 9ee86fa9ab0ab1..d85c8a6d338851 100644 --- a/src/coreclr/jit/codegenriscv64.cpp +++ b/src/coreclr/jit/codegenriscv64.cpp @@ -2207,7 +2207,55 @@ void CodeGen::genLockedInstructions(GenTreeOp* treeNode) default: noway_assert(!"Unexpected treeNode->gtOper"); } - GetEmitter()->emitIns_R_R_R(ins, dataSize, targetReg, addrReg, dataReg); + if (!m_compiler->compOpportunisticallyDependsOn(InstructionSet_A)) + { + // Without the A extension the ISA has no atomic memory operation, so the + // only lowering available is a plain read/modify/write. That is correct + // only on a target with a single hart and no preemption, which is the + // condition under which a build may select an ISA without A. + // + // XCHG needs no scratch; the arithmetic and bitwise forms compute the new + // value in an internal register so the original stays live in targetReg. + // BuildNode in lsrariscv64 extends the address and data lifetimes so they + // are not reused across the sequence. + instruction insLoad = is4 ? INS_lw : INS_ld; + instruction insStore = is4 ? INS_sw : INS_sd; + if (treeNode->OperIs(GT_XCHG)) + { + if (targetReg != REG_ZERO) + { + GetEmitter()->emitIns_R_R_I(insLoad, dataSize, targetReg, addrReg, 0); + } + GetEmitter()->emitIns_R_R_I(insStore, dataSize, dataReg, addrReg, 0); + } + else + { + regNumber tmpReg = internalRegisters.GetSingle(treeNode); + regNumber valueReg = (targetReg != REG_ZERO) ? targetReg : tmpReg; + instruction insOp; + switch (treeNode->gtOper) + { + case GT_XADD: + insOp = is4 ? INS_addw : INS_add; + break; + case GT_XAND: + insOp = INS_and; + break; + case GT_XORR: + insOp = INS_or; + break; + default: + unreached(); + } + GetEmitter()->emitIns_R_R_I(insLoad, dataSize, valueReg, addrReg, 0); + GetEmitter()->emitIns_R_R_R(insOp, dataSize, tmpReg, valueReg, dataReg); + GetEmitter()->emitIns_R_R_I(insStore, dataSize, tmpReg, addrReg, 0); + } + } + else + { + GetEmitter()->emitIns_R_R_R(ins, dataSize, targetReg, addrReg, dataReg); + } if (targetReg != REG_ZERO) { @@ -2270,19 +2318,32 @@ void CodeGen::genCodeForCmpXchg(GenTreeCmpXchg* treeNode) // so mark the location register as a GC pointer until code generation for this node is finished. gcInfo.gcMarkRegPtrVal(loc, locOp->TypeGet()); - BasicBlock* retry = genCreateTempLabel(); - BasicBlock* fail = genCreateTempLabel(); + BasicBlock* fail = genCreateTempLabel(); emitter* e = GetEmitter(); emitAttr size = emitActualTypeSize(valOp); bool is4 = (size == EA_4BYTE); - genDefineTempLabel(retry); - e->emitIns_R_R_R(is4 ? INS_lr_w : INS_lr_d, size, target, loc, REG_R0); // load original value - e->emitIns_J_cond_la(INS_bne, fail, target, comparand); // fail if doesn’t match - e->emitIns_R_R_R(is4 ? INS_sc_w : INS_sc_d, size, storeErr, loc, val); // try to update - e->emitIns_J_cond_la(INS_bnez, retry, storeErr); // retry if update failed - genDefineTempLabel(fail); + if (!m_compiler->compOpportunisticallyDependsOn(InstructionSet_A)) + { + // No A extension, so no lr/sc reservation pair. Compare and swap without + // one: old = *loc; if (old == comparand) *loc = val; result = old. As + // above, this holds only on a single-hart target with no preemption. + e->emitIns_R_R_I(is4 ? INS_lw : INS_ld, size, target, loc, 0); + e->emitIns_J_cond_la(INS_bne, fail, target, comparand); + e->emitIns_R_R_I(is4 ? INS_sw : INS_sd, size, val, loc, 0); + genDefineTempLabel(fail); + } + else + { + BasicBlock* retry = genCreateTempLabel(); + genDefineTempLabel(retry); + e->emitIns_R_R_R(is4 ? INS_lr_w : INS_lr_d, size, target, loc, REG_R0); // load original value + e->emitIns_J_cond_la(INS_bne, fail, target, comparand); // fail if doesn't match + e->emitIns_R_R_R(is4 ? INS_sc_w : INS_sc_d, size, storeErr, loc, val); // try to update + e->emitIns_J_cond_la(INS_bnez, retry, storeErr); // retry if update failed + genDefineTempLabel(fail); + } gcInfo.gcMarkRegSetNpt(locOp->gtGetRegMask()); genProduceReg(treeNode); diff --git a/src/coreclr/jit/lsrariscv64.cpp b/src/coreclr/jit/lsrariscv64.cpp index 656d7288bb2f45..5ea3b9cc246ded 100644 --- a/src/coreclr/jit/lsrariscv64.cpp +++ b/src/coreclr/jit/lsrariscv64.cpp @@ -563,18 +563,40 @@ int LinearScan::BuildNode(GenTree* tree) GenTree* data = tree->gtGetOp2(); assert(!addr->isContained()); - srcCount = 1; - BuildUse(addr); + // Without the A extension genLockedInstructions expands this to a + // multi-instruction read/modify/write that reuses the address and data + // registers after the first instruction, so their lifetimes have to be + // extended past the def. The arithmetic and bitwise forms also need one + // scratch register for the new value. + const bool plainAtomic = !m_compiler->compOpportunisticallyDependsOn(InstructionSet_A); + + srcCount = 1; + RefPosition* addrUse = BuildUse(addr); + if (plainAtomic) + { + setDelayFree(addrUse); + } if (!data->isContained()) { srcCount++; - BuildUse(data); + RefPosition* dataUse = BuildUse(data); + if (plainAtomic) + { + setDelayFree(dataUse); + } } else { assert(data->IsIntegralConst(0)); } + if (plainAtomic && !tree->OperIs(GT_XCHG)) + { + buildInternalIntRegisterDefForNode(tree); + setInternalRegsDelayFree = true; + buildInternalRegisterUses(); + } + if (dstCount == 1) { BuildDef(tree); From 2dc1eb4257a777927d4e132d9d54569e2bd37ed6 Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 16 Sep 2026 16:11:56 +0100 Subject: [PATCH 15/16] [RISC-V] Detect the A extension from the ISA string, and bump the JIT/EE GUID The A-extension test was a substring match for "a" against the whole -march string, which is wrong in both directions. rv64gc has the extension, through the "g" shorthand for imafd, but contains no letter "a". rv64im_zba does not have it, but contains "a" in the name of Zba. So the default baseline linked libatomic it does not need, and a build that really lacks A would fail on the atomic-alignment warning under -Werror. Match on the single-letter part of the string only, after stripping the rv prefix and any multi-letter extensions, and treat "g" as implying A. Also generate a new JITEEVersionIdentifier: the new CorInfoHelpFunc entries shift the values of every helper after them, so a JIT and an EE from different sides of this change must not be considered compatible. Signed-off-by: Maxim Menshikov --- eng/common/cross/toolchain.cmake | 12 ++++++++++-- eng/native/configurecompiler.cmake | 13 ++++++++++++- src/coreclr/inc/jiteeversionguid.h | 10 +++++----- 3 files changed, 27 insertions(+), 8 deletions(-) diff --git a/eng/common/cross/toolchain.cmake b/eng/common/cross/toolchain.cmake index 4a4ef49aa930cf..530ab5aa860783 100644 --- a/eng/common/cross/toolchain.cmake +++ b/eng/common/cross/toolchain.cmake @@ -372,8 +372,16 @@ elseif(TARGET_ARCH_NAME STREQUAL "riscv64") # Without the A extension the compiler lowers C/C++ atomics to __atomic_* # calls instead of emitting lr/sc, and those live in libatomic. Not every link # in the tree passes -latomic on its own. - if(DEFINED CLR_CMAKE_RISCV64_MARCH AND NOT CLR_CMAKE_RISCV64_MARCH MATCHES "a") - add_toolchain_linker_flag("-latomic") + # + # Only the single-letter part of the ISA string counts: a multi-letter extension + # whose name contains "a" (Zba, for one) is not the A extension, and "g" is + # shorthand for imafd and so implies it. + if(DEFINED CLR_CMAKE_RISCV64_MARCH) + string(REGEX REPLACE "_.*$" "" _riscv_single_letter "${CLR_CMAKE_RISCV64_MARCH}") + string(REGEX REPLACE "^rv[0-9]+" "" _riscv_single_letter "${_riscv_single_letter}") + if(NOT _riscv_single_letter MATCHES "[ag]") + add_toolchain_linker_flag("-latomic") + endif() endif() # persist variables across multiple try_compile passes diff --git a/eng/native/configurecompiler.cmake b/eng/native/configurecompiler.cmake index a326d49a21675b..52fb966fa2d42a 100644 --- a/eng/native/configurecompiler.cmake +++ b/eng/native/configurecompiler.cmake @@ -948,9 +948,20 @@ if(CLR_CMAKE_HOST_UNIX_RISCV64) set(CLR_CMAKE_RISCV64_MABI "$ENV{CLR_CMAKE_RISCV64_MABI}") endif() + # Decide whether the ISA string selects the A extension. Only the single-letter + # part counts: a multi-letter extension whose name contains "a" (Zba, for one) + # is not the A extension, and "g" is shorthand for imafd and so implies it. + string(REGEX REPLACE "_.*$" "" _riscv_single_letter "${CLR_CMAKE_RISCV64_MARCH}") + string(REGEX REPLACE "^rv[0-9]+" "" _riscv_single_letter "${_riscv_single_letter}") + if(_riscv_single_letter MATCHES "[ag]") + set(_riscv_has_a ON) + else() + set(_riscv_has_a OFF) + endif() + # Without the A extension every __atomic_* call is "not lock-free" and clang # warns; the runtime builds with -Werror, which would make that fatal. - if(NOT CLR_CMAKE_RISCV64_MARCH MATCHES "a") + if(NOT _riscv_has_a) add_compile_options(-Wno-atomic-alignment) endif() diff --git a/src/coreclr/inc/jiteeversionguid.h b/src/coreclr/inc/jiteeversionguid.h index fb33de9eaa0023..8f00452418b9f8 100644 --- a/src/coreclr/inc/jiteeversionguid.h +++ b/src/coreclr/inc/jiteeversionguid.h @@ -37,11 +37,11 @@ #include -constexpr GUID JITEEVersionIdentifier = { /* fa0c6a6f-b219-4b60-b928-042c72667eb3 */ - 0xfa0c6a6f, - 0xb219, - 0x4b60, - {0xb9, 0x28, 0x04, 0x2c, 0x72, 0x66, 0x7e, 0xb3} +constexpr GUID JITEEVersionIdentifier = { /* a3daece5-c930-442e-886d-0cf93f4d6e54 */ + 0xa3daece5, + 0xc930, + 0x442e, + {0x88, 0x6d, 0x0c, 0xf9, 0x3f, 0x4d, 0x6e, 0x54} }; #endif // JIT_EE_VERSIONING_GUID_H From 353000e9709677bdc2a7d73a4775617fa5e3103d Mon Sep 17 00:00:00 2001 From: Maxim Menshikov Date: Wed, 16 Sep 2026 18:20:37 +0100 Subject: [PATCH 16/16] [RISC-V] Require --assume-no-concurrency to build without the A extension Without A there is no atomic memory operation to emit, so Interlocked and CmpXchg lower to a plain read/modify/write. Whether that is sufficient depends on the target running on one hart and never being preempted. Neither the ISA nor the ABI says anything about that: a soft-float lp64 image is perfectly normal on rv64imac hardware with threads, and would lose atomicity silently. So it has to be asserted rather than inferred. Add --assume-no-concurrency to ilc and reject an ISA without A unless it is passed. The name states the whole promise - one hart and no preemption - rather than only the first half, and the failure is at compile time rather than at run time. Signed-off-by: Maxim Menshikov --- src/coreclr/tools/Common/InstructionSetHelpers.cs | 9 +++++---- .../tools/aot/ILCompiler/ILCompilerRootCommand.cs | 3 +++ src/coreclr/tools/aot/ILCompiler/Program.cs | 12 ++++++++++++ 3 files changed, 20 insertions(+), 4 deletions(-) diff --git a/src/coreclr/tools/Common/InstructionSetHelpers.cs b/src/coreclr/tools/Common/InstructionSetHelpers.cs index 7c4d16241d3e9f..5240af5130fdae 100644 --- a/src/coreclr/tools/Common/InstructionSetHelpers.cs +++ b/src/coreclr/tools/Common/InstructionSetHelpers.cs @@ -96,10 +96,11 @@ public static InstructionSetSupport ConfigureInstructionSetSupport(string instru } else if (targetArchitecture == TargetArchitecture.RiscV64) { - // The rv64gc baseline: D implies F, so "d", "c" and "a" cover - // the G+C extensions. Reduced-ISA targets (e.g. zkVM guests) - // opt out with --instruction-set=-a,-c,-d,-f. The lp64 (soft-float) - // ABI target has no F/D by definition. + // The rv64gc baseline: D implies F, so "d", "c" and "a" cover the G+C + // extensions. The lp64 (soft-float) ABI target has no F/D by definition; + // it still defaults to C and A, and a reduced-ISA target drops those with + // --instruction-set=-a,-c. Dropping A also requires ilc's + // --assume-no-concurrency, because Interlocked is not atomic without it. instructionSetSupportBuilder.AddSupportedInstructionSet("base"); if (targetAbi != TargetAbi.NativeAotRiscV64SoftFloat) { diff --git a/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs b/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs index 9bed4be8a61018..f2e47cf6012b1a 100644 --- a/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs +++ b/src/coreclr/tools/aot/ILCompiler/ILCompilerRootCommand.cs @@ -118,6 +118,8 @@ internal sealed class ILCompilerRootCommand : RootCommand new("--parallelism") { CustomParser = MakeParallelism, DefaultValueFactory = MakeParallelism, Description = "Maximum number of threads to use during compilation" }; public Option InstructionSet { get; } = new("--instruction-set") { Description = "Instruction set to allow or disallow" }; + public Option AssumeNoConcurrency { get; } = + new("--assume-no-concurrency") { Description = "RISC-V: assert that the target runs on one hart and is never preempted. Required to build without the A extension, because Interlocked then lowers to a non-atomic read/modify/write" }; public Option MaxVectorTBitWidth { get; } = new("--max-vectort-bitwidth") { Description = "Maximum width, in bits, that Vector is allowed to be" }; public Option Guard { get; } = @@ -243,6 +245,7 @@ public ILCompilerRootCommand(string[] args) : base(".NET Native IL Compiler") Options.Add(RuntimeKnobs); Options.Add(Parallelism); Options.Add(InstructionSet); + Options.Add(AssumeNoConcurrency); Options.Add(MaxVectorTBitWidth); Options.Add(Guard); Options.Add(Dehydrate); diff --git a/src/coreclr/tools/aot/ILCompiler/Program.cs b/src/coreclr/tools/aot/ILCompiler/Program.cs index 86aac8ced8393d..4dab289a37afda 100644 --- a/src/coreclr/tools/aot/ILCompiler/Program.cs +++ b/src/coreclr/tools/aot/ILCompiler/Program.cs @@ -112,6 +112,18 @@ public int Run() isReadyToRun: false, targetAbi: targetAbi); + if (targetArchitecture == TargetArchitecture.RiscV64 && + !instructionSetSupport.IsInstructionSetSupported(InstructionSet.RiscV64_A) && + !Get(_command.AssumeNoConcurrency)) + { + // Without A there is no atomic memory operation to emit, so Interlocked and + // CmpXchg lower to a plain read/modify/write. Whether that is sufficient is a + // property of the execution environment, not of the ISA or the ABI, so it has + // to be asserted rather than inferred. + throw new CommandLineException( + "Building without the A extension requires --assume-no-concurrency (one hart, no preemption): Interlocked operations are not atomic without it."); + } + string systemModuleName = Get(_command.SystemModuleName); string reflectionData = Get(_command.ReflectionData); bool supportsReflection = reflectionData != "none" && systemModuleName == Helpers.DefaultSystemModule;