diff --git a/models/cpu/iss/flexfloat/flexfloat.c b/models/cpu/iss/flexfloat/flexfloat.c index 1d7900e63..bce181e55 100644 --- a/models/cpu/iss/flexfloat/flexfloat.c +++ b/models/cpu/iss/flexfloat/flexfloat.c @@ -183,7 +183,11 @@ bool flexfloat_sticky_bit(const flexfloat_t *a, int_fast16_t exp) ( ((denorm & MASK_FRAC) == 0) && (CAST_TO_INT(a->value)!=0) ); #else unsigned short shift = NUM_BITS_FRAC - a->desc.frac_bits - exp + 1; - if (shift>=NUM_BITS) return 0; + // Every significant bit (the value is normal in the backend, hence + // nonzero) sits below the sticky window: the sticky bit is set. A + // zero here would lose inexact/underflow and directed rounding to + // the smallest subnormal for deeply tiny values. + if (shift>=NUM_BITS) return 1; uint_t Frac = (CAST_TO_INT(a->value) & MASK_FRAC) | MASK_FRAC_MSB; int StiB = ((Frac & (((uint_t)1<<(shift-1))-1)) != 0); return (StiB!=0); @@ -241,14 +245,41 @@ int_t flexfloat_rounding_value(const flexfloat_t *a, int_fast16_t exp, bool sign } +// apply a one-ulp (target grid) increment to the backend value +static void flexfloat_apply_rounding(flexfloat_t *a, int_fast16_t exp, bool sign) +{ + int_t rounding_value = flexfloat_rounding_value(a, exp, sign); + /* Truncate the discarded bits first (integer, exact): both addends are + * then on the target grid and the sum is exact. Adding the ulp to the + * raw backend value rounds in the backend format and can overshoot the + * target grid by one ulp when the discarded bits extend to the bottom + * of the backend mantissa. */ + if (EXPONENT(CAST_TO_INT(a->value)) != 0) + { + int shift = NUM_BITS_FRAC - a->desc.frac_bits + ((exp <= 0) ? - exp + 1 : 0); + if (shift <= NUM_BITS_FRAC) + CAST_TO_INT(a->value) &= ~((UINT_C(1) << shift) - UINT_C(1)); + else // magnitude entirely below one target ulp: truncate to (signed) zero + CAST_TO_INT(a->value) &= UINT_C(1) << (NUM_BITS - 1); + } + a->value += CAST_TO_FP(rounding_value); +} + #endif // FLEXFLOAT_ROUNDING +// RISC-V RMM support: with FE_TONEAREST set, a tie (round bit 1, sticky 0) is +// resolved away from zero instead of to even. See flexfloat.h. +int flexfloat_rmm = 0; + void flexfloat_sanitize(flexfloat_t *a) { bool sign; int_fast16_t exp; int_fast16_t inf_exp; uint_t frac; +#ifdef FLEXFLOAT_FLAGS + bool inexact = false; +#endif if (isnan(a->value)) { @@ -268,33 +299,72 @@ void flexfloat_sanitize(flexfloat_t *a) exp = flexfloat_exp(a); #ifdef FLEXFLOAT_ROUNDING - // In these cases no rounding is needed - if (!(exp == INF_EXP || a->desc.frac_bits == NUM_BITS_FRAC)) + /* No rounding needed for a propagated Inf/NaN or a full-width target. + * Also skip it when the target-format exponent already overflows + * (exp >= inf_exp of the target): rounding cannot bring the exponent + * back into range, and the overflow branch below owns the delivery + * (OF|NX per IEEE 754 7.4, mode-aware clamp vs infinity). Rounding + * here would be worse than useless: at exactly + * exp - frac_bits == inf_exp the increment built by + * flexfloat_rounding_value packs as the target's infinity + * (flexfloat_pack maps exp == inf_exp to INF_EXP), the += turns the + * value into Inf mid-rounding with its flags deliberately discarded, + * and the Inf branch below then reads it as an infinity propagated + * from an operand - delivering Inf with the overflow flag lost. */ + if (!(exp == INF_EXP || exp >= flexfloat_inf_exp(a->desc) + || a->desc.frac_bits == NUM_BITS_FRAC)) { #ifdef FLEXFLOAT_FLAGS // Inexact results raise an exception - if(flexfloat_round_bit(a, exp) || flexfloat_sticky_bit(a, exp)) + inexact = flexfloat_round_bit(a, exp) || flexfloat_sticky_bit(a, exp); + if(inexact) feraiseexcept(FE_INEXACT); + if (inexact && exp <= 0 && EXPONENT(CAST_TO_INT(a->value)) != 0) + { + /* Underflow = tiny AND inexact (IEEE 754 §7.5). */ + bool escapes = false; +#ifdef FLEXFLOAT_TININESS_AFTER_ROUNDING + /* IEEE 754 leaves the tininess detection point implementation- + * defined. This opt-in per-core c-flag selects detection after + * rounding with unbounded exponent range: a subnormal-range + * value escapes tininess only when, at full target precision, + * it rounds up across the smallest normal (probed with the + * helpers' normal branch, exp arg 1). Default: the historical + * before-rounding detection. */ + if (exp == 0) + { + uint_t frac_all = (UINT_C(1) << a->desc.frac_bits) - 1; + bool all_ones = (((CAST_TO_INT(a->value) & MASK_FRAC) + >> (NUM_BITS_FRAC - a->desc.frac_bits)) == frac_all); + int m = fegetround(); + bool up = (m == FE_TONEAREST && (flexfloat_rmm ? flexfloat_round_bit(a, 1) + : flexfloat_nearest_rounding(a, 1))) + || (m == FE_UPWARD && flexfloat_inf_rounding(a, 1, sign, 1)) + || (m == FE_DOWNWARD && flexfloat_inf_rounding(a, 1, sign, 0)); + escapes = all_ones && up; + } +#endif + if (!escapes) + feraiseexcept(FE_UNDERFLOW); + } // As rounding uses FP operations, we don't want to tarnish the accrued flags fexcept_t flags; fegetexceptflag(&flags, FE_ALL_EXCEPT); #endif // Rounding mode int mode = fegetround(); - if(mode == FE_TONEAREST && flexfloat_nearest_rounding(a, exp)) + if(mode == FE_TONEAREST && (flexfloat_rmm ? flexfloat_round_bit(a, exp) + : flexfloat_nearest_rounding(a, exp))) { - int_t rounding_value = flexfloat_rounding_value(a, exp, sign); - a->value += CAST_TO_FP(rounding_value); + flexfloat_apply_rounding(a, exp, sign); } else if(mode == FE_UPWARD && flexfloat_inf_rounding(a, exp, sign, 1)) { - int_t rounding_value = flexfloat_rounding_value(a, exp, sign); - a->value += CAST_TO_FP(rounding_value); + flexfloat_apply_rounding(a, exp, sign); } else if(mode == FE_DOWNWARD && flexfloat_inf_rounding(a, exp, sign, 0)) { - int_t rounding_value = flexfloat_rounding_value(a, exp, sign); - a->value += CAST_TO_FP(rounding_value); + flexfloat_apply_rounding(a, exp, sign); } #ifdef FLEXFLOAT_FLAGS // Restore flags from before @@ -322,8 +392,11 @@ void flexfloat_sanitize(flexfloat_t *a) if(exp <= 0) // Denormalized value in the target format (saved in normalized format in the backend value) { -#ifdef FLEXFLOAT_FLAGS - // Raise the underflow exception +#if defined(FLEXFLOAT_FLAGS) && !defined(FLEXFLOAT_ROUNDING) + /* With rounding enabled, underflow is raised in the rounding block + * above (tiny and inexact, tininess after rounding with unbounded + * exponent range). Keep the legacy unconditional raise for + * flag-only builds. */ feraiseexcept(FE_UNDERFLOW); #endif uint_t denorm = flexfloat_denorm_frac(a, exp); @@ -355,21 +428,38 @@ void flexfloat_sanitize(flexfloat_t *a) } else if(exp == INF_EXP) // Inf { -#ifdef FLEXFLOAT_FLAGS - // Raise the proper overflow exception, unless a DIV/0 exception had occured - if (!fetestexcept(FE_DIVBYZERO)) - feraiseexcept(FE_OVERFLOW | FE_INEXACT); -#endif + /* No overflow here. IEEE 754 7.4 raises OF only when the result + * rounded with an unbounded exponent range exceeds the destination + * range; a backend value that is already Inf is an infinity + * propagated from an operand (or produced by a division by zero, + * which raises its own flag) and is exact. A genuine destination + * overflow leaves the backend finite and is handled by the + * (exp >= inf_exp) branch below. */ exp = inf_exp; } - else if(exp >= inf_exp) // Out of bounds for target format: set infinity + else if(exp >= inf_exp) // Out of bounds for target format: overflow { #ifdef FLEXFLOAT_FLAGS // Raise the proper overflow exception feraiseexcept(FE_OVERFLOW | FE_INEXACT); #endif - exp = inf_exp; - frac = UINT_C(0); + // IEEE 754 overflow: directed roundings that point away from the + // overflow direction clamp to the largest finite value instead of + // infinity (toward-zero always; downward for positive, upward for + // negative). Nearest (even or ties-away) overflows to infinity. + int ovf_mode = fegetround(); + if (ovf_mode == FE_TOWARDZERO || + (ovf_mode == FE_DOWNWARD && !sign) || + (ovf_mode == FE_UPWARD && sign)) + { + exp = inf_exp - 1; + frac = (UINT_C(1) << a->desc.frac_bits) - 1; + } + else + { + exp = inf_exp; + frac = UINT_C(0); + } } // printf("ENCODING: %d %d %lu\n", sign, exp, frac); @@ -645,34 +735,56 @@ INLINE void ff_fma(flexfloat_t *dest, const flexfloat_t *a, const flexfloat_t *b assert((dest->desc.exp_bits == a->desc.exp_bits) && (dest->desc.frac_bits == a->desc.frac_bits) && (a->desc.exp_bits == b->desc.exp_bits) && (a->desc.frac_bits == b->desc.frac_bits) && (b->desc.exp_bits == c->desc.exp_bits) && (b->desc.frac_bits == c->desc.frac_bits)); + #ifdef FLEXFLOAT_FLAGS + /* inf*0 signals INVALID even when the addend is a quiet NaN: IEEE 754 + * 7.2 leaves that sub-case implementation-defined and the host FMA + * propagates the NaN silently, but RISC-V mandates the flag (unpriv F). + * Idempotent when the addend is not a NaN: the host raises it too. */ + if ((a->value == 0.0 && isinf(b->value)) || + (isinf(a->value) && b->value == 0.0)) + feraiseexcept(FE_INVALID); + #endif #ifdef FLEXFLOAT_ROUNDING - // Change the rounding mode according to the error direction if we need to do manual rounding for RNE - int mode = fegetround(); - bool eff_sub = flexfloat_sign(a) ^ flexfloat_sign(b) ^ flexfloat_sign(c); - if (a->desc.frac_bits < NUM_BITS_FRAC && mode == FE_TONEAREST) { - if (!eff_sub) { // in this case, we need to round away from zero - fexcept_t flags; - fegetexceptflag(&flags, FE_ALL_EXCEPT); // get accrued flags to not tarnish them here - double try = fma(a->value, b->value, c->value); - (try >= 0) ? fesetround(FE_UPWARD) : fesetround(FE_DOWNWARD); - fesetexceptflag(&flags, FE_ALL_EXCEPT); // restore flags here - } else { -#ifdef OLD - fesetround(FE_TOWARDZERO); // just truncate -#endif - } + if (a->desc.frac_bits < NUM_BITS_FRAC) + { + /* Round-to-odd intermediate (Boldo-Muller): truncate the fused op + * toward zero and, if inexact, force the mantissa LSB to 1. An odd + * binary64 intermediate keeps faithful round AND sticky information + * for any narrower destination, so the software rounding in + * flexfloat_sanitize (every mode, RMM included) yields the + * single-rounding fused result - the double-rounding artefact is + * structurally gone. The binary64 op cannot overflow or + * underflow on narrow-format inputs; INVALID is re-raised, its + * INEXACT is dropped - sanitize re-derives the destination + * inexactness from the odd/round/sticky bits. */ + fexcept_t accrued; + fegetexceptflag(&accrued, FE_ALL_EXCEPT); + feclearexcept(FE_ALL_EXCEPT); + int mode = fegetround(); + fesetround(FE_TOWARDZERO); + double r = fma(a->value, b->value, c->value); + int raised = fetestexcept(FE_ALL_EXCEPT); + fesetround(mode); + fesetexceptflag(&accrued, FE_ALL_EXCEPT); + if (raised & FE_INVALID) + feraiseexcept(FE_INVALID); + if ((raised & FE_INEXACT) && !isnan(r) && !isinf(r)) + CAST_TO_INT(r) |= 1; + else if (r == 0.0) + /* Exact zero: its sign depends on the actual rounding mode + * (-0 on exact cancellation under RDN), which the toward-zero + * run hides. Recompute exactly, no flags raised. */ + r = fma(a->value, b->value, c->value); + dest->value = r; } + else #endif - dest->value = fma(a->value, b->value, c->value); // finally the actual operation + dest->value = fma(a->value, b->value, c->value); // full-width destination: plain fused op #ifdef FLEXFLOAT_TRACKING dest->exact_value = fma(a->exact_value, b->exact_value, c->exact_value); if(dest->tracking_fn) (dest->tracking_fn)(dest, dest->tracking_arg); #endif - #ifdef FLEXFLOAT_ROUNDING - if (a->desc.frac_bits < NUM_BITS_FRAC && mode == FE_TONEAREST) - fesetround(FE_TONEAREST); // restore rounding - #endif flexfloat_sanitize(dest); #ifdef FLEXFLOAT_STATS if(StatsEnabled) getOpStats(dest->desc)->fma += 1; @@ -683,33 +795,53 @@ INLINE void ff_fnma(flexfloat_t *dest, const flexfloat_t *a, const flexfloat_t * assert((dest->desc.exp_bits == a->desc.exp_bits) && (dest->desc.frac_bits == a->desc.frac_bits) && (a->desc.exp_bits == b->desc.exp_bits) && (a->desc.frac_bits == b->desc.frac_bits) && (b->desc.exp_bits == c->desc.exp_bits) && (b->desc.frac_bits == c->desc.frac_bits)); + #ifdef FLEXFLOAT_FLAGS + /* inf*0 signals INVALID even when the addend is a quiet NaN: IEEE 754 + * 7.2 leaves that sub-case implementation-defined and the host FMA + * propagates the NaN silently, but RISC-V mandates the flag (unpriv F). + * Idempotent when the addend is not a NaN: the host raises it too. */ + if ((a->value == 0.0 && isinf(b->value)) || + (isinf(a->value) && b->value == 0.0)) + feraiseexcept(FE_INVALID); + #endif #ifdef FLEXFLOAT_ROUNDING - // Change the rounding mode according to the error direction if we need to do manual rounding for RNE - int mode = fegetround(); - bool eff_sub = flexfloat_sign(a) ^ flexfloat_sign(b) ^ flexfloat_sign(c); - if (a->desc.frac_bits < NUM_BITS_FRAC && mode == FE_TONEAREST) { - if (!eff_sub) { // in this case, we need to round away from zero - fexcept_t flags; - fegetexceptflag(&flags, FE_ALL_EXCEPT); // get accrued flags to not tarnish them here - double try = fma(a->value, b->value, c->value); - (try >= 0) ? fesetround(FE_UPWARD) : fesetround(FE_DOWNWARD); - fesetexceptflag(&flags, FE_ALL_EXCEPT); // restore flags here - } else { - fesetround(FE_TOWARDZERO); // just truncate + if (a->desc.frac_bits < NUM_BITS_FRAC) + { + /* Round-to-odd intermediate, then negate: truncation toward zero + * and the odd bit are sign-symmetric, so -round_to_odd(a*b+c) is + * the odd-rounded -(a*b+c). See ff_fma for the full rationale. */ + fexcept_t accrued; + fegetexceptflag(&accrued, FE_ALL_EXCEPT); + feclearexcept(FE_ALL_EXCEPT); + int mode = fegetround(); + fesetround(FE_TOWARDZERO); + double r = fma(a->value, b->value, c->value); + int raised = fetestexcept(FE_ALL_EXCEPT); + fesetround(mode); + fesetexceptflag(&accrued, FE_ALL_EXCEPT); + if (raised & FE_INVALID) + feraiseexcept(FE_INVALID); + if ((raised & FE_INEXACT) && !isnan(r) && !isinf(r)) + { + CAST_TO_INT(r) |= 1; + dest->value = -r; } + else if (r == 0.0) + /* Exact zero: the sign must be derived on the negated fusion + * in the actual rounding mode (-0 on exact cancellation under + * RDN) - negating the toward-zero result would flip it. */ + dest->value = fma(-a->value, b->value, -c->value); + else + dest->value = -r; } + else #endif - dest->value = -fma(a->value, b->value, c->value); #ifdef FLEXFLOAT_TRACKING dest->exact_value = fma(a->exact_value, b->exact_value, c->exact_value); if(dest->tracking_fn) (dest->tracking_fn)(dest, dest->tracking_arg); #endif - #ifdef FLEXFLOAT_ROUNDING - if (a->desc.frac_bits < NUM_BITS_FRAC && mode == FE_TONEAREST) - fesetround(FE_TONEAREST); // restore rounding - #endif flexfloat_sanitize(dest); #ifdef FLEXFLOAT_STATS if(StatsEnabled) getOpStats(dest->desc)->fma += 1; diff --git a/models/cpu/iss/flexfloat/flexfloat.h b/models/cpu/iss/flexfloat/flexfloat.h index e7791f101..76357ff89 100644 --- a/models/cpu/iss/flexfloat/flexfloat.h +++ b/models/cpu/iss/flexfloat/flexfloat.h @@ -195,6 +195,11 @@ uint_t flexfloat_denorm_frac(const flexfloat_t *a, int_fast16_t exp); uint_t flexfloat_pack(flexfloat_desc_t desc, bool sign, int_fast16_t exp, uint_t frac); void flexfloat_sanitize(flexfloat_t *a); +// Round-to-nearest, ties away from zero (RISC-V RMM). The C fenv has no such +// mode: callers set FE_TONEAREST plus this flag, and flexfloat_sanitize then +// resolves ties away from zero instead of to even. +extern int flexfloat_rmm; + // Bit-level access diff --git a/models/cpu/iss/include/core.hpp b/models/cpu/iss/include/core.hpp index a47b7cdc8..36067e88d 100644 --- a/models/cpu/iss/include/core.hpp +++ b/models/cpu/iss/include/core.hpp @@ -32,6 +32,9 @@ class Core { public: Core(Iss &iss); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual ~Core() = default; +#endif void build(); void reset(bool active); @@ -40,6 +43,14 @@ class Core iss_reg_t dret_handle(); iss_reg_t sret_handle(); int mode_get() { return this->mode; } +#ifdef CONFIG_GVSOC_ISS_CV32E40P + /* Privilege-mode restore on MRET. Default: from mstatus.mpp. */ + virtual void mret_mode_restore(); + + /* mstatus_write_mask customization, called at end of build() after the + * generic mask is computed. Default: no-op. */ + virtual void mstatus_write_mask_fixup() {} +#endif void mode_set(int mode); iss_reg_t load_reserve_addr_get() { return this->load_reserve_addr; } void load_reserve_addr_set(iss_reg_t addr) { this->load_reserve_addr = addr; } @@ -50,15 +61,24 @@ class Core vp::Trace event_jal; vp::Trace event_jalr; +#ifdef CONFIG_GVSOC_ISS_CV32E40P +protected: + Iss &iss; + iss_reg_t mstatus_write_mask; +#endif private: bool mstatus_update(bool is_write, iss_reg_t &value); bool sstatus_update(bool is_write, iss_reg_t &value); +#ifndef CONFIG_GVSOC_ISS_CV32E40P Iss &iss; +#endif vp::Trace trace; int mode; +#ifndef CONFIG_GVSOC_ISS_CV32E40P iss_reg_t mstatus_write_mask; +#endif iss_reg_t sstatus_write_mask; iss_reg_t load_reserve_addr; bool reset_stall = false; diff --git a/models/cpu/iss/include/cores/cv32e40p/__init__.py b/models/cpu/iss/include/cores/cv32e40p/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/models/cpu/iss/include/cores/cv32e40p/class.hpp b/models/cpu/iss/include/cores/cv32e40p/class.hpp new file mode 100644 index 000000000..f77536cfd --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/class.hpp @@ -0,0 +1,134 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + + +#pragma once + + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#ifdef CONFIG_GVSOC_ISS_RISCV_EXCEPTIONS +#include +#else +#include +#endif +#include +#if defined(CONFIG_GVSOC_ISS_MMU) +#include +#endif +#if defined(CONFIG_GVSOC_ISS_PMP) +#include +#endif +#include +#include +#include +#include +#include + +class IssWrapper; + + +class Iss +{ +public: + Iss(IssWrapper &top); + + Cv32e40pRegfile regfile; + Exec exec; + InsnCache insn_cache; + Timing timing; + Cv32e40pCore core; + Prefetcher prefetcher; + Decode decode; + Cv32e40pIrq irq; + Gdbserver gdbserver; + Lsu lsu; + DbgUnit dbgunit; + Syscalls syscalls; + Trace trace; + Cv32e40pCsr csr; +#if defined(CONFIG_GVSOC_ISS_MMU) + Mmu mmu; +#endif +#if defined(CONFIG_GVSOC_ISS_PMP) + Pmp pmp; +#endif + Cv32e40pException exception; + Memcheck memcheck; + + vp::Component ⊤ +}; + +class IssWrapper : public vp::Component +{ + +public: + IssWrapper(vp::ComponentConf &config); + + void start(); + void stop(); + void reset(bool active); + + Iss iss; + +private: + vp::Trace trace; +}; + +inline Iss::Iss(IssWrapper &top) + : prefetcher(*this), exec(top, *this), insn_cache(*this), decode(*this), timing(*this), core(*this), irq(*this), + gdbserver(*this), lsu(top, *this), dbgunit(*this), syscalls(top, *this), trace(*this), csr(*this), + regfile(top, *this), exception(*this), memcheck(top, *this), top(top) +#if defined(CONFIG_GVSOC_ISS_MMU) + , mmu(*this) +#endif +#if defined(CONFIG_GVSOC_ISS_PMP) + , pmp(*this) +#endif +{ +} + +#include "cpu/iss/include/isa/rv64i.hpp" +#include "cpu/iss/include/isa/rv32i.hpp" +#include "cpu/iss/include/isa/rv32c.hpp" +#include "cpu/iss/include/isa/zcmp.hpp" +#include "cpu/iss/include/isa/rv32a.hpp" +#include "cpu/iss/include/isa/rv64c.hpp" +#include "cpu/iss/include/isa/rv32m.hpp" +#include "cpu/iss/include/isa/rv64m.hpp" +#include "cpu/iss/include/isa/rv64a.hpp" +#include "cpu/iss/include/isa/rvf.hpp" +#include "cpu/iss/include/isa/rvd.hpp" +#include "cpu/iss/include/cores/cv32e40p/priv.hpp" +#include + + +#include diff --git a/models/cpu/iss/include/cores/cv32e40p/core.hpp b/models/cpu/iss/include/cores/cv32e40p/core.hpp new file mode 100644 index 000000000..0c55921a8 --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/core.hpp @@ -0,0 +1,42 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#pragma once + +#include + +class Cv32e40pCore : public Core +{ +public: + Cv32e40pCore(Iss &iss) : Core(iss) {} + +protected: + /* CV32E40P is M-mode only. + * CSR writes can set MPP to 0, but + * MRET must ignore that and stay in M-mode. */ + void mret_mode_restore() override; + + /* CV32E40P mstatus_write_mask FPU-aware. + * only MIE(3) + MPIE(7) writable; MPP forced to M by hardware. + * With FPU: add FS(14:13). */ + void mstatus_write_mask_fixup() override; +}; diff --git a/models/cpu/iss/include/cores/cv32e40p/csr.hpp b/models/cpu/iss/include/cores/cv32e40p/csr.hpp new file mode 100644 index 000000000..683ed8230 --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/csr.hpp @@ -0,0 +1,73 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#pragma once + +#include + +class Cv32e40pCsr : public Csr +{ +public: + Cv32e40pCsr(Iss &iss); + + void build(); // Calls Csr::build() then build_cv32e40p() + void build_cv32e40p(); // CV32E40P-specific CSR customization + void reset(bool active); + + bool mstatus_access(bool is_write, iss_reg_t &value) override; + bool mcycle_access(bool is_write, iss_reg_t &value) override; + void mstatus_read_fixup(iss_reg_t &value) override; + + // FP/Vector CSR access pre-check. + // Returns true (illegal) when mstatus[FS] == 00 (Off) + // raises illegal-instruction on FP CSR access while FS=Off. + bool fp_access_illegal() override; + + // Promote mstatus.FS to Dirty(11) on any FP state change. + // RTL (cv32e40p_cs_registers.sv) forces FS=Dirty when FPU=1 && ZFINX=0 on + // FP regfile write, fflags update, or FP-CSR write. SD(bit31) is derived + // on read (SD = FS==3), not stored here. + void fp_state_dirty() override; + + // CoreV2 HWLOOP CSR mapping. + // 0xCC0..0xCC2 → 0..2 (lpstart0/lpend0/lpcount0) + // 0xCC4..0xCC6 → 4..6 (lpstart1/lpend1/lpcount1) + // gap at 0xCC3 / outside range → -1. + int hwloop_csr_index(iss_reg_t reg) override; + + // CoreV2 HWLOOP CSR names for trace messages. + const char *custom_csr_name(iss_reg_t reg) override; + + // EBREAK in M-mode enters debug when dcsr.ebreakm=1. + // RISC-V Debug Spec: bit 15 of dcsr is ebreakm. + bool ebreak_m_mode_enters_debug() override; + + // PULP custom CSRs (0xCD0-0xCD2) + CsrReg uhartid; // 0xCD0 — duplicate of mhartid (user-mode readable) + CsrReg privlv; // 0xCD1 — current privilege level + CsrReg zfinx_csr; // 0xCD2 — ZFINX indicator; undeclared (illegal) when FPU=1 && ZFINX=0 + +private: + int64_t mcycle_offset = 0; + bool m_fpu_in_isa = false; + bool m_zfinx = false; +}; diff --git a/models/cpu/iss/include/cores/cv32e40p/exception.hpp b/models/cpu/iss/include/cores/cv32e40p/exception.hpp new file mode 100644 index 000000000..b1cc2b8d5 --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/exception.hpp @@ -0,0 +1,36 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#pragma once + +#include + +class Cv32e40pException : public Exception +{ +public: + /* CV32E40P always aligns the trap vector to 4 bytes + set base-class alignment mask accordingly. */ + Cv32e40pException(Iss &iss) : Exception(iss) + { + this->trap_vector_align_mask = ~(iss_reg_t)0x3; + } +}; diff --git a/models/cpu/iss/include/cores/cv32e40p/irq.hpp b/models/cpu/iss/include/cores/cv32e40p/irq.hpp new file mode 100644 index 000000000..2d2179c36 --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/irq.hpp @@ -0,0 +1,40 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#pragma once + +#include + +class Cv32e40pIrq : public Irq +{ +public: + Cv32e40pIrq(Iss &iss) : Irq(iss) {} + + bool mip_access(bool is_write, iss_reg_t &value) override; + bool mie_access(bool is_write, iss_reg_t &value) override; + bool mtvec_access(bool is_write, iss_reg_t &value) override; + void elw_irq_unstall() override; + iss_reg_t compute_trap_entry(iss_reg_t base, int cause, bool is_interrupt) override; + +protected: + void register_csr_callbacks() override; +}; diff --git a/models/cpu/iss/include/cores/cv32e40p/priv.hpp b/models/cpu/iss/include/cores/cv32e40p/priv.hpp new file mode 100644 index 000000000..7ea1cc39b --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/priv.hpp @@ -0,0 +1,274 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +/* + * CV32E40P-specific override of isa/priv.hpp. + * + * Fixes CSRRC/CSRRCI/CSRRSI to NOT treat rs1=x0 / uimm=0 as a write, per + * RISC-V Privileged Spec §2.2: + * "If rs1=x0, then the instruction shall not write to the CSR at all." + * "For CSRRSI/CSRRCI, if uimm=0, the instruction shall not write the CSR." + * + * csrrs_exec is already correct upstream and is reproduced verbatim below. + */ + +#pragma once + +#include + +static inline void csr_decode(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // In case traces are active, convert the CSR number into a name +#ifdef VP_TRACE_ACTIVE + insn->args[2].flags = (iss_decoder_arg_flag_e)(insn->args[2].flags | ISS_DECODER_ARG_FLAG_DUMP_NAME); + insn->args[2].name = iss_csr_name(iss, UIM_GET(0)); +#endif +} + +static inline iss_reg_t csrrw_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr) + { + return csr->handle(iss, insn, pc, REG_GET(0)); + } + + iss_reg_t value; + iss_reg_t reg_value = REG_GET(0); + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + // For now we don't have any mechanism to track validity of CSR, so set output + // register as valid + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + iss_csr_write(iss, insn, UIM_GET(0), reg_value); + + return iss_insn_next(iss, insn, pc); +} + +/* + * csrrc_exec — CV32E40P fix: CSRRC with rs1=x0 must NOT write the CSR. + * RISC-V Priv Spec §2.2: if rs1=x0 the instruction shall not write to the CSR. + */ +static inline iss_reg_t csrrc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + iss_reg_t reg_value = REG_GET(0); + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, REG_IN(0) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + // For now we don't have any mechanism to track validity of CSR, so set output + // register as valid + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (REG_IN(0) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value & ~reg_value); + } + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + iss_reg_t reg_value = REG_GET(0); + + #ifdef CONFIG_GVSOC_ISS_SNITCH + // Todo: put csr mcycle performance couter value assignment from here to csr.cpp + if (UIM_GET(0) == 0xB00) + { + iss->csr.mcycle.value = iss->top.clock.get_cycles(); + } + #endif + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, REG_IN(0) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + // For now we don't have any mechanism to track validity of CSR, so set output + // register as valid + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + if (REG_IN(0) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value | reg_value); + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrwi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr) + { + return csr->handle(iss, insn, pc, UIM_GET(1)); + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + // For now we don't have any mechanism to track validity of CSR, so set output + // register as valid + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + iss_csr_write(iss, insn, UIM_GET(0), UIM_GET(1)); + return iss_insn_next(iss, insn, pc); +} + +/* + * csrrci_exec — CV32E40P fix: CSRRCI with uimm=0 must NOT write the CSR. + * RISC-V Priv Spec §2.2: for CSRRCI, if uimm[4:0]=0 the instruction shall not write the CSR. + */ +static inline iss_reg_t csrrci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, UIM_GET(1) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + // For now we don't have any mechanism to track validity of CSR, so set output + // register as valid + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (UIM_GET(1) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value & ~UIM_GET(1)); + } + + return iss_insn_next(iss, insn, pc); +} + +/* + * csrrsi_exec — CV32E40P fix: CSRRSI with uimm=0 must NOT write the CSR. + * RISC-V Priv Spec §2.2: for CSRRSI, if uimm[4:0]=0 the instruction shall not write the CSR. + */ +static inline iss_reg_t csrrsi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, UIM_GET(1) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + // For now we don't have any mechanism to track validity of CSR, so set output + // register as valid + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (UIM_GET(1) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value | UIM_GET(1)); + } + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t wfi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->irq.wfi_handle(); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t mret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->exec.irq_exit.set(1); + iss->timing.stall_insn_dependency_account(5); + return iss->core.mret_handle(); +} + +static inline iss_reg_t dret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + return iss->core.dret_handle(); +} + +static inline iss_reg_t sret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->timing.stall_insn_dependency_account(5); + return iss->core.sret_handle(); +} + +static inline iss_reg_t sfence_vma_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->core.mode_get() == PRIV_S && iss->csr.mstatus.tvm) + { + iss->exception.raise(pc, ISS_EXCEPT_ILLEGAL); + return pc; + } + else + { +#ifdef CONFIG_GVSOC_ISS_MMU + iss->mmu.flush(REG_GET(0), REG_GET(1)); +#endif + iss->insn_cache.mode_flush(); + return iss_insn_next(iss, insn, pc); + } +} diff --git a/models/cpu/iss/include/cores/cv32e40p/regfile.hpp b/models/cpu/iss/include/cores/cv32e40p/regfile.hpp new file mode 100644 index 000000000..f2b35ea36 --- /dev/null +++ b/models/cpu/iss/include/cores/cv32e40p/regfile.hpp @@ -0,0 +1,39 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + + +#pragma once + +#include + + +class Cv32e40pRegfile : public Regfile +{ +public: + Cv32e40pRegfile(IssWrapper &top, Iss &iss) : Regfile(top, iss) {} + + void reset(bool active) { + Regfile::reset(active); + for (int i = 0; i < ISS_NB_REGS; i++) this->regs[i] = 0; + } + +}; diff --git a/models/cpu/iss/include/csr.hpp b/models/cpu/iss/include/csr.hpp index 8f9598753..59383b4a1 100644 --- a/models/cpu/iss/include/csr.hpp +++ b/models/cpu/iss/include/csr.hpp @@ -52,12 +52,23 @@ class CsrAbtractReg const char *name; iss_reg_t reset_val; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + iss_reg_t write_mask; +#endif bool write_illegal = false; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + + // Public setter for write_mask — allows core-specific subclasses to + // configure CSR masks without requiring friend access. + void set_write_mask(iss_reg_t mask) { write_mask = mask; } +#endif protected: void reset(bool active); +#ifndef CONFIG_GVSOC_ISS_CV32E40P iss_reg_t write_mask; +#endif private: std::vector> callbacks; @@ -152,6 +163,9 @@ class Csr void build(); void reset(bool active); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual ~Csr() = default; +#endif void declare_pcer(int index, std::string name, std::string help); void declare_csr(CsrAbtractReg *reg, std::string name, iss_reg_t address, iss_reg_t reset_val=0, iss_reg_t mask=-1); @@ -214,6 +228,22 @@ class Csr #endif CsrReg mcountinhibit; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Optional extension CSRs (CV32E40P-only). + // When the config is off, access falls through to the "unsupported CSR" warning. +#if ISS_REG_WIDTH == 32 + CsrReg mcycleh; + CsrReg minstreth; +#endif + + CsrReg minstret; + CsrReg mhpmevent[29]; + + CsrReg tinfo; + CsrReg mcontext; + CsrReg scontext; + +#endif /* CONFIG_GVSOC_ISS_CV32E40P */ #if defined(CONFIG_GVSOC_ISS_PMP) CsrReg pmpcfg[16]; CsrReg pmpaddr[64]; @@ -236,7 +266,12 @@ class Csr iss_reg_t scratch0; iss_reg_t scratch1; iss_fcsr_t fcsr; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + CsrReg mhartid; + CsrReg mimpid; +#else iss_reg_t mhartid; +#endif CsrReg vstart; CsrReg vxstat; @@ -250,6 +285,66 @@ class Csr iss_reg_t hwloop_regs[HWLOOP_NB_REGS]; #endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P + /* Hook for core-specific mstatus read fixup (e.g. SD bit). + * Called by Core::mstatus_update on read path. Default: no-op. */ + virtual void mstatus_read_fixup(iss_reg_t &value) {} + + /* FP/Vector CSR access pre-check. + * Default: allow (returns false → not illegal). + * Called from static fflags/frm/fcsr read/write to detect illegal access. */ + virtual bool fp_access_illegal() { return false; } + + /* Promote mstatus.FS to Dirty on FP state change. + * Default: no-op → preserves behavior for all non-CV32E40P cores + * (Snitch/PULP/etc.). CV32E40P RTL forces FS=Dirty on FP regfile write, + * fflags update, or FP-CSR write. Override in Cv32e40pCsr. */ + virtual void fp_state_dirty() {} + + /* Default tselect read value when no trigger is selected. + * Default: -1 (all-1s, conventional "no trigger" sentinel). */ + iss_reg_t tselect_default_read = (iss_reg_t)-1; + + /* Behavior on access to an undeclared/unsupported CSR. + * Default: false → log warning, no exception (legacy GVSOC behavior). */ + bool raise_on_unsupported_csr_flag = false; + + /* Map a CSR address to its core-specific HWLOOP register index. + * Default: -1 (not a HWLOOP CSR). + * Used by iss_csr_read (dispatch to hwloop_read) and iss_csr_write. */ + virtual int hwloop_csr_index(iss_reg_t reg) { return -1; } + + /* Core-specific CSR name lookup for trace messages. + * Default: nullptr (fall through to generic table). */ + virtual const char *custom_csr_name(iss_reg_t reg) { return nullptr; } + + /* Whether Exec::bootaddr_apply derives mtvec from boot address. + * Default: true (generic RISC-V: mtvec = bootaddr & ~0xFF). */ + bool bootaddr_writes_mtvec_flag = true; + + /* EBREAK in M-mode behavior. + * Default: false → raise ISS_EXCEPT_BREAKPOINT (generic RISC-V w/o debug). */ + virtual bool ebreak_m_mode_enters_debug() { return false; } + +protected: + /* mstatus access hook. + * Default: pass through (return true). */ + virtual bool mstatus_access(bool is_write, iss_reg_t &value); + /* mcycle access hook. + * Default: on read, return current clock cycle count; on write, no-op. */ + virtual bool mcycle_access(bool is_write, iss_reg_t &value); + void undeclare_csr(iss_reg_t address) { regs.erase(address); } + + std::map regs; + +private: + + bool tselect_access(bool is_write, iss_reg_t &value); + bool time_access(bool is_write, iss_reg_t &value); + vp::WireMaster time_itf; + +}; +#else private: bool tselect_access(bool is_write, iss_reg_t &value); @@ -260,3 +355,4 @@ class Csr vp::WireMaster time_itf; }; +#endif diff --git a/models/cpu/iss/include/exception.hpp b/models/cpu/iss/include/exception.hpp index 1e217ab15..5ad47c6e3 100644 --- a/models/cpu/iss/include/exception.hpp +++ b/models/cpu/iss/include/exception.hpp @@ -28,6 +28,9 @@ class Exception { public: Exception(Iss &iss); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual ~Exception() = default; +#endif void build(); @@ -35,7 +38,17 @@ class Exception iss_addr_t debug_handler_addr; +#ifdef CONFIG_GVSOC_ISS_CV32E40P +protected: +#else private: +#endif Iss &iss; vp::Trace trace; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + + /* Trap vector PC alignment mask. + * Default: -1 (no masking — generic RISC-V). */ + iss_reg_t trap_vector_align_mask = (iss_reg_t)-1; +#endif }; diff --git a/models/cpu/iss/include/irq/irq_riscv.hpp b/models/cpu/iss/include/irq/irq_riscv.hpp index 53220a16c..a4e60c9c1 100644 --- a/models/cpu/iss/include/irq/irq_riscv.hpp +++ b/models/cpu/iss/include/irq/irq_riscv.hpp @@ -44,15 +44,27 @@ class Irq { public: Irq(Iss &iss); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual ~Irq() = default; +#endif void build(); bool mideleg_access(bool is_write, iss_reg_t &value); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual bool mip_access(bool is_write, iss_reg_t &value); + virtual bool mie_access(bool is_write, iss_reg_t &value); +#else bool mip_access(bool is_write, iss_reg_t &value); bool mie_access(bool is_write, iss_reg_t &value); +#endif bool sip_access(bool is_write, iss_reg_t &value); bool sie_access(bool is_write, iss_reg_t &value); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual bool mtvec_access(bool is_write, iss_reg_t &value); +#else bool mtvec_access(bool is_write, iss_reg_t &value); +#endif bool stvec_access(bool is_write, iss_reg_t &value); bool mtvec_set(iss_addr_t base); @@ -61,9 +73,28 @@ class Irq void cache_flush(); void reset(bool active); int check(); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // CV32E40P: compute the interrupt trap-vector entry. Default returns the + // base unchanged (= upstream direct-mode mtvec.value); Cv32e40pIrq overrides + // it to add vectored mode (mtvec.MODE=1 -> base + cause*4). + virtual iss_reg_t compute_trap_entry(iss_reg_t base, int cause, bool is_interrupt); +#endif void wfi_handle(); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + virtual void elw_irq_unstall(); +#else void elw_irq_unstall(); +#endif void check_interrupts(); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + +protected: + /* Register CSR access callbacks (mip/mie/mtvec). + * Default: registers handlers for mip, mie, mtvec on the base Csr class. */ + virtual void register_csr_callbacks(); + +public: +#endif static void msi_sync(vp::Block *__this, bool value); static void mti_sync(vp::Block *__this, bool value); static void mei_sync(vp::Block *__this, bool value); diff --git a/models/cpu/iss/include/isa/corev.hpp b/models/cpu/iss/include/isa/corev.hpp index ea5f5144c..e667a6af9 100644 --- a/models/cpu/iss/include/isa/corev.hpp +++ b/models/cpu/iss/include/isa/corev.hpp @@ -23,28 +23,51 @@ #ifndef __CPU_ISS_COREV_HPP #define __CPU_ISS_COREV_HPP +/* Self-sufficient like rv32i.hpp: iss_v2 emits ISA-subset includes in + * subset order, so this header cannot rely on rv32i.hpp having pulled + * in the macros before it. */ +#ifdef CONFIG_GVSOC_ISS_V2 +#include "cpu/iss/include/isa_lib/int.h" +#include "cpu/iss_v2/include/isa_lib/macros.h" +#else +#include "cpu/iss/include/iss_core.hpp" +#include "cpu/iss/include/isa_lib/int.h" +#include "cpu/iss/include/isa_lib/macros.h" +#endif + #define COREV_HWLOOP_LPSTART0 0 #define COREV_HWLOOP_LPEND0 1 #define COREV_HWLOOP_LPCOUNT0 2 -#define COREV_HWLOOP_LPSTART1 3 -#define COREV_HWLOOP_LPEND1 4 -#define COREV_HWLOOP_LPCOUNT1 5 +#define COREV_HWLOOP_LPSTART1 4 +#define COREV_HWLOOP_LPEND1 5 +#define COREV_HWLOOP_LPCOUNT1 6 -#define COREV_HWLOOP_LPSTART(x) (COREV_HWLOOP_LPSTART0 + (x)*3) -#define COREV_HWLOOP_LPEND(x) (COREV_HWLOOP_LPEND0 + (x)*3) -#define COREV_HWLOOP_LPCOUNT(x) (COREV_HWLOOP_LPCOUNT0 + (x)*3) +#define COREV_HWLOOP_LPSTART(x) (COREV_HWLOOP_LPSTART0 + (x)*4) +#define COREV_HWLOOP_LPEND(x) (COREV_HWLOOP_LPEND0 + (x)*4) +#define COREV_HWLOOP_LPCOUNT(x) (COREV_HWLOOP_LPCOUNT0 + (x)*4) +/* v1-only hardware-loop machinery: the stub handler and the exec/csr-side + * state it drives do not exist on iss_v2, where the dispatch loop calls + * iss->hwloop.check() natively and the lp_* handlers below program the + * Hwloop module directly. */ +#ifndef CONFIG_GVSOC_ISS_V2 static inline iss_reg_t corev_hwloop_check_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { // Check now is the instruction has been replayed to know if it is the first // time it is executed bool elw_interrupted = iss->exec.elw_interrupted; - // First execute the instructions as it is the last one of the loop body. + // First execute the instruction as it is the last one of the loop body. // The real handler has been saved when the loop was started. iss_reg_t insn_next = insn->hwloop_handler(iss, insn, pc); + // If halted (e.g. by elw stall), do not process hwloop logic + if (iss->exec.halted.get()) + { + return insn_next; + } + if (elw_interrupted) { // This flag is 1 when the instruction has been previously interrupted and is now @@ -55,7 +78,8 @@ static inline iss_reg_t corev_hwloop_check_exec(Iss *iss, iss_insn_t *insn, iss_ } // First check HW loop 0 as it has higher priority compared to HW loop 1 - if (iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT0] && iss->csr.hwloop_regs[COREV_HWLOOP_LPEND0] == pc) + // CV32E40P: the loop-end stub fires at LPEND-4, so stored LPEND == pc + 4. + if (iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT0] && iss->csr.hwloop_regs[COREV_HWLOOP_LPEND0] == pc + 4) { iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT0]--; iss->decode.trace.msg("Reached end of HW loop (index: 0, loop count: %d)\n", iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT0]); @@ -71,7 +95,8 @@ static inline iss_reg_t corev_hwloop_check_exec(Iss *iss, iss_insn_t *insn, iss_ // We get here either if HW loop 0 was not active or if the counter reached 0. // In both cases, HW loop 1 can jump back to the beginning of the loop. - if (iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT1] && iss->csr.hwloop_regs[COREV_HWLOOP_LPEND1] == pc) + // CV32E40P: the loop-end stub fires at LPEND-4, so stored LPEND == pc + 4. + if (iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT1] && iss->csr.hwloop_regs[COREV_HWLOOP_LPEND1] == pc + 4) { iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT1]--; // If counter is not zero, we must jump back to beginning of the loop. @@ -87,628 +112,1989 @@ static inline iss_reg_t corev_hwloop_check_exec(Iss *iss, iss_insn_t *insn, iss_ // In case no HW loop jumped back, just continue with the next instruction. iss->exec.hwloop_next_insn = insn_next; - return insn_next; + return insn_next; +} + +static inline void corev_hwloop_set_start(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start) +{ + // lpstart holds a word address; bits [1:0] are hardwired 0. Align the + // register-sourced operand (the csrw path is masked in hwloop_write). + start &= ~0x3u; + iss->csr.hwloop_regs[COREV_HWLOOP_LPSTART(index)] = start; + iss->exec.hwloop_set_start(index, start); +} + +static inline void corev_hwloop_set_end(Iss *iss, iss_insn_t *insn, int index, iss_reg_t end) +{ + // lpend holds a word address; bits [1:0] are hardwired 0. + end &= ~0x3u; + iss->csr.hwloop_regs[COREV_HWLOOP_LPEND(index)] = end; + // CV32E40P loops back from the last body instruction (LPEND-4), not LPEND. + // The last body instruction is always 32-bit, so the offset is exactly 4. + iss->exec.hwloop_set_end(index, end - 4); +} + +static inline void corev_hwloop_set_count(Iss *iss, iss_insn_t *insn, int index, iss_reg_t count) +{ + iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT(index)] = count; +} + +static inline void corev_hwloop_set_all(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start, iss_reg_t end, iss_reg_t count) +{ + corev_hwloop_set_end(iss, insn, index, end); + corev_hwloop_set_start(iss, insn, index, start); + corev_hwloop_set_count(iss, insn, index, count); +} +#else +/* iss_v2 versions of the CV32E40P hwloop setters, programming the Hwloop + * module directly. Same RTL-grounded semantics as the v1 path: lpstart/lpend + * bits [1:0] are hardwired 0, and the core loops back from the LAST BODY + * instruction, i.e. the module's end must match LPEND - 4 (the v2 check() + * compares the just-executed pc against the stored end). The architectural + * LPEND is kept in the csr personality (hwloop_lpend), which is also the + * CSR read path: re-deriving it from the module as get_end() + 4 would + * read back 4 instead of 0 on a never-programmed loop. */ +static inline void corev_hwloop_set_start(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start) +{ + iss->hwloop.set_start(index, start & ~(iss_reg_t)0x3); +} + +static inline void corev_hwloop_set_end(Iss *iss, iss_insn_t *insn, int index, iss_reg_t end) +{ + iss->csr.hwloop_lpend[index] = end & ~(iss_reg_t)0x3; + iss->hwloop.set_end(index, (end & ~(iss_reg_t)0x3) - 4); +} + +static inline void corev_hwloop_set_count(Iss *iss, iss_insn_t *insn, int index, iss_reg_t count) +{ + iss->hwloop.set_count(index, count); +} + +static inline void corev_hwloop_set_all(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start, iss_reg_t end, iss_reg_t count) +{ + corev_hwloop_set_end(iss, insn, index, end); + corev_hwloop_set_start(iss, insn, index, start); + corev_hwloop_set_count(iss, insn, index, count); +} +#endif + +/* Dead cv_* scalar handlers removed — the decoder (isa_cv32e40pv2.py) uses + * p_* names from the PulpV2 section below. The cv_* versions were unused + * duplicates (L= parameter only sets trace label, not handler name). */ + +/* iss_handle_elw from pulp_v2.hpp — needed by cv_elw_exec and p_elw_exec */ +#ifndef CONFIG_GVSOC_ISS_V2 +static inline void iss_handle_elw(Iss *iss, iss_insn_t *insn, iss_reg_t pc, iss_addr_t addr, int size, int reg) +{ + // Always account the overhead of the elw + iss->timing.stall_insn_account(2); + + iss->exec.elw_insn = pc; + // Init this flag so that we can check afterwards that theelw has been replayed + iss->exec.elw_interrupted = 0; + + iss->lsu.elw_perf(insn, addr, size, reg); + + if (!iss->exec.stalled.get()) + { + } + else + { + // Since an interrupt might have happened during the execution of the elw, we need to check them + // in case this is waking-up the elw. + if (iss->irq.check()) + { + iss->irq.elw_irq_unstall(); + } + else + { + iss->lsu.elw_stalled.set(true); + iss->exec.busy_exit(); + } + } +} +#endif + + +/* Dead code removed: cv_elw_exec (decoder uses p_elw_exec), + * CV_OP_*_EXEC SIMD macros (CV32E40P has no SIMD v1), + * cv_extractr/insertr/shuffle/dot/pack handlers (no decoder refs), + * cv_mac/msu/mul/add/sub norm handlers (decoder uses p_* versions). */ + +static inline iss_reg_t cv_clipr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // Spec (instruction_set_extensions.rst:786): rs2' = rs2 & 0x7FFFFFFF + int high = REG_GET(1) & 0x7fffffff; + int low = -high - 1; + REG_SET(0, LIB_CALL3(lib_CLIP, REG_GET(0), low, high)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_clipur_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // Spec (instruction_set_extensions.rst:794): rs2' = rs2 & 0x7FFFFFFF + int low = 0; + int high = REG_GET(1) & 0x7fffffff; + REG_SET(0, LIB_CALL3(lib_CLIP, REG_GET(0), low, high)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_bclr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = UIM_GET(0) + 1; + int shift = UIM_GET(1); + REG_SET(0, LIB_CALL2(lib_BCLR, REG_GET(0), ((1ULL << width) - 1) << shift)); + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_bclrr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = ((REG_GET(1) >> 5) & 0x1f) + 1; + int shift = REG_GET(1) & 0x1f; + REG_SET(0, LIB_CALL2(lib_BCLR, REG_GET(0), ((1ULL << width) - 1) << shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_extract_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = UIM_GET(0) + 1; + int shift = UIM_GET(1); + REG_SET(0, LIB_CALL3(lib_BEXTRACT, REG_GET(0), width, shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_extractu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = UIM_GET(0) + 1; + int shift = UIM_GET(1); + REG_SET(0, LIB_CALL3(lib_BEXTRACTU, REG_GET(0), ((1ULL << width) - 1) << shift, shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_extractr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = ((REG_GET(1) >> 5) & 0x1f) + 1; + int shift = REG_GET(1) & 0x1f; + REG_SET(0, LIB_CALL3(lib_BEXTRACT, REG_GET(0), width, shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_extractur_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = ((REG_GET(1) >> 5) & 0x1f) + 1; + int shift = REG_GET(1) & 0x1f; + REG_SET(0, LIB_CALL3(lib_BEXTRACTU, REG_GET(0), ((1ULL << width) - 1) << shift, shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_insert_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = UIM_GET(0) + 1; + int shift = UIM_GET(1); + REG_SET(0, LIB_CALL4(lib_BINSERT, REG_GET(0), REG_GET(1), ((1ULL << width) - 1) << shift, shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_insertr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = ((REG_GET(2) >> 5) & 0x1F) + 1; + int shift = REG_GET(2) & 0x1F; + REG_SET(0, LIB_CALL4(lib_BINSERT, REG_GET(0), REG_GET(1), ((1ULL << width) - 1) << shift, shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_bset_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = UIM_GET(0) + 1; + int shift = UIM_GET(1); + REG_SET(0, LIB_CALL2(lib_BSET, REG_GET(0), ((1ULL << (width)) - 1) << shift)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_bsetr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + int width = ((REG_GET(1) >> 5) & 0x1f) + 1; + int shift = REG_GET(1) & 0x1f; + REG_SET(0, LIB_CALL2(lib_BSET, REG_GET(0), ((1ULL << (width)) - 1) << shift)); + return iss_insn_next(iss, insn, pc); +} + +// CV32E40P-exact variants of the generic XPULP scalar helpers, used where the +// shared lib diverges from the RTL (cv.bitrev, cv.machhuRN/cv.macuRN). +static inline uint32_t rev32_cv32e40p(uint32_t x) +{ + uint32_t r = 0; + for (int i = 0; i < 32; i++) if ((x >> i) & 1) r |= 1u << (31 - i); + return r; +} +// cv.bitrev: reverse(rs1), shift right by Is2, reverse back, then radix-reverse +// selected by Is3 (0 = full radix-2 reverse, 1 = radix-4 2-bit groups, 2 = radix-8 +// 3-bit groups). The shared lib_BITREV uses a different group-rotation algorithm. +static inline unsigned int lib_BITREV_cv32e40p(Iss *s, unsigned int input, unsigned int Is2, unsigned int Is3) +{ + uint32_t sh = rev32_cv32e40p(rev32_cv32e40p(input) >> (Is2 & 0x1f)); + uint32_t sel = Is3 & 0x3; + uint32_t out = 0; + if (sel == 1) { for (int j = 0; j < 16; j++) out |= ((sh >> (31 - 2 * j - 1)) & 0x3u) << (2 * j); } + else if (sel == 2) { for (int j = 0; j < 10; j++) out |= ((sh >> (31 - 3 * j - 2)) & 0x7u) << (3 * j); } + else { out = rev32_cv32e40p(sh); } + return out; +} +// CV32E40P-exact unsigned MAC high/low-half round-and-normalize (cv.machhuRN / +// cv.macuRN). The accumulate is kept in 32 bits, so the rounding-add carry past bit +// 31 is dropped before the shift; the shared lib promotes it to 64 bits, which lets +// the carry survive and corrupt large shifts. +static inline unsigned int lib_MAC_ZH_ZH_NR_R_cv32e40p(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) +{ + uint32_t result = (uint32_t)(a + ((b >> 16) & 0xffffu) * ((c >> 16) & 0xffffu)); + if (shift > 0) result = (uint32_t)(result + (1u << (shift - 1))) >> shift; + return result; +} +static inline unsigned int lib_MAC_ZL_ZL_NR_R_cv32e40p(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) +{ + uint32_t result = (uint32_t)(a + (b & 0xffffu) * (c & 0xffffu)); + if (shift > 0) result = (uint32_t)(result + (1u << (shift - 1))) >> shift; + return result; +} + +static inline iss_reg_t cv_bitrev_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + REG_SET(0, LIB_CALL3(lib_BITREV_cv32e40p, REG_GET(0), UIM_GET(0), UIM_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +/* p.bitrev label alias — isa_gen derives handler name from label, + * so p.bitrev -> p_bitrev_exec; forward to the canonical cv_bitrev_exec. */ +static inline iss_reg_t p_bitrev_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + return cv_bitrev_exec(iss, insn, pc); +} + + +/* lp.start / lp.end: register-based hwloop setup (CV32E40P-specific). + * The immediate versions (lp_starti_exec, lp_endi_exec) are in pulp_v2 section below. + * These use REG_GET(0) as the address value instead of PC-relative immediate. */ +static inline iss_reg_t lp_start_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + corev_hwloop_set_start(iss, insn, UIM_GET(0), REG_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t lp_end_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + corev_hwloop_set_end(iss, insn, UIM_GET(0), REG_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +/* ================================================================ + * PULP V2 handlers — ported for CV32E40P independence. + * ================================================================ */ + +#define PULPV2_HWLOOP_LPSTART0 COREV_HWLOOP_LPSTART0 +#define PULPV2_HWLOOP_LPEND0 COREV_HWLOOP_LPEND0 +#define PULPV2_HWLOOP_LPCOUNT0 COREV_HWLOOP_LPCOUNT0 +#define PULPV2_HWLOOP_LPSTART1 COREV_HWLOOP_LPSTART1 +#define PULPV2_HWLOOP_LPEND1 COREV_HWLOOP_LPEND1 +#define PULPV2_HWLOOP_LPCOUNT1 COREV_HWLOOP_LPCOUNT1 +#define PULPV2_HWLOOP_LPSTART(x) (PULPV2_HWLOOP_LPSTART0 + (x)*4) +#define PULPV2_HWLOOP_LPEND(x) (PULPV2_HWLOOP_LPEND0 + (x)*4) +#define PULPV2_HWLOOP_LPCOUNT(x) (PULPV2_HWLOOP_LPCOUNT0 + (x)*4) + +static inline iss_reg_t LB_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.load_signed(insn, REG_GET(0) + REG_GET(1), 1, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LB_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(1)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(1)); + iss->lsu.stack_access_check(REG_IN(1), REG_GET(0) + REG_GET(1)); + if (iss->lsu.load_signed_perf(insn, REG_GET(0) + REG_GET(1), 1, REG_OUT(0))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LH_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.load_signed(insn, REG_GET(0) + REG_GET(1), 2, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LH_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(1)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(1)); + iss->lsu.stack_access_check(REG_IN(1), REG_GET(0) + REG_GET(1)); + if (iss->lsu.load_signed_perf(insn, REG_GET(0) + REG_GET(1), 2, REG_OUT(0))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LW_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.load(insn, REG_GET(0) + REG_GET(1), 4, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LW_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(1)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(1)); + iss->lsu.stack_access_check(REG_IN(1), REG_GET(0) + REG_GET(1)); + if (iss->lsu.load_perf(insn, REG_GET(0) + REG_GET(1), 4, REG_OUT(0))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LBU_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.load(insn, REG_GET(0) + REG_GET(1), 1, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LBU_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(1)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(1)); + iss->lsu.stack_access_check(REG_IN(1), REG_GET(0) + REG_GET(1)); + if (iss->lsu.load_perf(insn, REG_GET(0) + REG_GET(1), 1, REG_OUT(0))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LHU_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.load(insn, REG_GET(0) + REG_GET(1), 2, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LHU_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(1)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(1)); + iss->lsu.stack_access_check(REG_IN(1), REG_GET(0) + REG_GET(1)); + if (iss->lsu.load_perf(insn, REG_GET(0) + REG_GET(1), 2, REG_OUT(0))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LB_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + if (iss->lsu.load_signed(insn, base, 1, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + /* rd==rs1: loaded data has priority over the incremented address (instruction_set_extensions.rst) */ + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LB_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss_reg_t base = REG_GET(0); + if (iss->lsu.load_signed_perf(insn, base, 1, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LH_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + if (iss->lsu.load_signed(insn, base, 2, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LH_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss_reg_t base = REG_GET(0); + if (iss->lsu.load_signed_perf(insn, base, 2, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LW_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + if (iss->lsu.load_signed(insn, base, 4, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LW_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss_reg_t base = REG_GET(0); + if (iss->lsu.load_signed_perf(insn, base, 4, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LBU_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + if (iss->lsu.load(insn, base, 1, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LBU_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss_reg_t base = REG_GET(0); + iss->lsu.stack_access_check(REG_IN(0), base); + if (iss->lsu.load_perf(insn, base, 1, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LHU_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + if (iss->lsu.load(insn, base, 2, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LHU_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss_reg_t base = REG_GET(0); + iss->lsu.stack_access_check(REG_IN(0), base); + if (iss->lsu.load_perf(insn, base, 2, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SB_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.store(insn, REG_GET(0), 1, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + IN_REG_SET(0, REG_GET(0) + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SB_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0)); + if (iss->lsu.store_perf(insn, REG_GET(0), 1, REG_IN(1))) + { + return pc; + } + IN_REG_SET(0, REG_GET(0) + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SH_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.store(insn, REG_GET(0), 2, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + IN_REG_SET(0, REG_GET(0) + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SH_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0)); + if (iss->lsu.store_perf(insn, REG_GET(0), 2, REG_IN(1))) + { + return pc; + } + IN_REG_SET(0, REG_GET(0) + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SW_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.store(insn, REG_GET(0), 4, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + IN_REG_SET(0, REG_GET(0) + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SW_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0)); + if (iss->lsu.store_perf(insn, REG_GET(0), 4, REG_IN(1))) + { + return pc; + } + IN_REG_SET(0, REG_GET(0) + SIM_GET(0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LB_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load_signed(insn, base, 1, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LB_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(1)); + + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load_signed_perf(insn, base, 1, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LH_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load_signed(insn, base, 2, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LH_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(1)); + + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load_signed_perf(insn, base, 2, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LW_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load_signed(insn, base, 4, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LW_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(1)); + + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load_signed_perf(insn, base, 4, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LBU_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load(insn, base, 1, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LBU_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(1)); + + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + iss->lsu.stack_access_check(REG_IN(0), base); + if (iss->lsu.load_perf(insn, base, 1, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LHU_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + if (iss->lsu.load(insn, base, 2, REG_OUT(0))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t LHU_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(1)); + + iss_reg_t base = REG_GET(0); + iss_reg_t offset = REG_GET(1); + iss->lsu.stack_access_check(REG_IN(0), base); + if (iss->lsu.load_perf(insn, base, 2, REG_OUT(0))) + { + return pc; + } + if (REG_OUT(0) != REG_IN(0)) + IN_REG_SET(0, base + offset); + return iss_insn_next(iss, insn, pc); } -static inline void corev_hwloop_set_start(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start) +static inline iss_reg_t SB_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - iss->csr.hwloop_regs[COREV_HWLOOP_LPSTART(index)] = start; - iss->exec.hwloop_start_insn[index] = start; + iss_reg_t new_val = REG_GET(0) + REG_GET(2); + if (iss->lsu.store(insn, REG_GET(0), 1, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + IN_REG_SET(0, new_val); + return iss_insn_next(iss, insn, pc); } -static inline void corev_hwloop_set_insn_end(Iss *iss, iss_reg_t pc) +static inline iss_reg_t SB_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - // TODO INSN - abort(); - // if (insn->fetched) - // { - // if (insn->hwloop_handler == NULL) - // { - // insn->hwloop_handler = insn->handler; - // insn->handler = hwloop_check_exec; - // insn->fast_handler = hwloop_check_exec; - // } - // } - // else - // { - // insn->hwloop_handler = hwloop_check_exec; - // } + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(2)); + + iss_reg_t new_val = REG_GET(0) + REG_GET(2); + iss->lsu.stack_access_check(REG_OUT(0), REG_GET(0)); + if (iss->lsu.store_perf(insn, REG_GET(0), 1, REG_IN(1))) + { + return pc; + } + IN_REG_SET(0, new_val); + return iss_insn_next(iss, insn, pc); } -static inline void corev_hwloop_set_end(Iss *iss, iss_insn_t *insn, int index, iss_reg_t end) +static inline iss_reg_t SH_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - iss->exec.hwloop_end_insn[index] = end; + iss_reg_t new_val = REG_GET(0) + REG_GET(2); + if (iss->lsu.store(insn, REG_GET(0), 2, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + IN_REG_SET(0, new_val); + return iss_insn_next(iss, insn, pc); +} - corev_hwloop_set_insn_end(iss, end); +static inline iss_reg_t SH_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(2)); - iss->csr.hwloop_regs[COREV_HWLOOP_LPEND(index)] = end; + iss_reg_t new_val = REG_GET(0) + REG_GET(2); + iss->lsu.stack_access_check(REG_OUT(0), REG_GET(0)); + if (iss->lsu.store_perf(insn, REG_GET(0), 2, REG_IN(1))) + { + return pc; + } + IN_REG_SET(0, new_val); + return iss_insn_next(iss, insn, pc); } -static inline void corev_hwloop_set_count(Iss *iss, iss_insn_t *insn, int index, iss_reg_t count) +static inline iss_reg_t SW_RR_POSTINC_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - iss->csr.hwloop_regs[COREV_HWLOOP_LPCOUNT(index)] = count; + iss_reg_t new_val = REG_GET(0) + REG_GET(2); + if (iss->lsu.store(insn, REG_GET(0), 4, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + IN_REG_SET(0, new_val); + return iss_insn_next(iss, insn, pc); } -static inline void corev_hwloop_set_all(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start, iss_reg_t end, iss_reg_t count) +static inline iss_reg_t SW_RR_POSTINC_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - corev_hwloop_set_end(iss, insn, index, end); - corev_hwloop_set_start(iss, insn, index, start); - corev_hwloop_set_count(iss, insn, index, count); + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + // Since input register is incremented, whole register becomes invalid if any bit is invalid + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_IN(0), REG_IN(2)); + + iss_reg_t new_val = REG_GET(0) + REG_GET(2); + iss->lsu.stack_access_check(REG_OUT(0), REG_GET(0)); + if (iss->lsu.store_perf(insn, REG_GET(0), 4, REG_IN(1))) + { + return pc; + } + IN_REG_SET(0, new_val); + return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_avgu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_avgu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL2(lib_AVGU, REG_GET(0), REG_GET(1))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_slet_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_slet_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, (int32_t)REG_GET(0) <= (int32_t)REG_GET(1)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_sletu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_sletu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, REG_GET(0) <= REG_GET(1)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_min_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_min_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_MINS, REG_GET(0), REG_GET(1))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_minu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_minu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_MINU, REG_GET(0), REG_GET(1))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_max_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_max_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_MAXS, REG_GET(0), REG_GET(1))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_maxu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_maxu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_MAXU, REG_GET(0), REG_GET(1))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_ror_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_ror_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + if (!iss->regfile.memcheck_get_valid(REG_IN(1))) + { + // If register containing the rotation is invalid, this is making the whole output + // register invalid + iss->regfile.memcheck_set_valid(REG_OUT(0), false); + } + else + { + // Otherwise, handle the bits separately + iss->regfile.memcheck_set(REG_OUT(0), + LIB_CALL2(lib_ROR, iss->regfile.memcheck_get(REG_IN(0)), REG_GET(1))); + } + REG_SET(0, LIB_CALL2(lib_ROR, REG_GET(0), REG_GET(1))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_ff1_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_ff1_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL1(lib_FF1, REG_GET(0))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_fl1_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_fl1_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL1(lib_FL1, REG_GET(0))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_clb_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_clb_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL1(lib_CLB, REG_GET(0))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_cnt_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_cnt_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL1(lib_CNT, REG_GET(0))); - // setRegDelayed(cpu, pc->outReg[0], value, 2); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_exths_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_exths_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_set(REG_OUT(0), + iss_get_signed_value(iss->regfile.memcheck_get(REG_IN(0)), 16)); + REG_SET(0, iss_get_signed_value(REG_GET(0), 16)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_exthz_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_exthz_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_set(REG_OUT(0), + iss_get_field(iss->regfile.memcheck_get(REG_IN(0)), 0, 16) | + (((1 << (ISS_REG_WIDTH - 16)) - 1) << 16)); + REG_SET(0, iss_get_field(REG_GET(0), 0, 16)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extbs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_extbs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_set(REG_OUT(0), + iss_get_signed_value(iss->regfile.memcheck_get(REG_IN(0)), 8)); + REG_SET(0, iss_get_signed_value(REG_GET(0), 8)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extbz_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_extbz_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_set(REG_OUT(0), + iss_get_field(iss->regfile.memcheck_get(REG_IN(0)), 0, 8) | + (((1 << (ISS_REG_WIDTH - 8)) - 1) << 8)); + REG_SET(0, iss_get_field(REG_GET(0), 0, 8)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_starti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +/* Unified hwloop functions — PulpV2 names delegate to corev_ versions. + * PULPV2_HWLOOP_* are aliases of COREV_HWLOOP_* (same register layout), + * so the implementations are identical. */ +#ifndef CONFIG_GVSOC_ISS_SNITCH_PULP_V2 +#ifndef CONFIG_GVSOC_ISS_V2 +static inline iss_reg_t hwloop_check_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + return corev_hwloop_check_exec(iss, insn, pc); +} +#endif + +static inline void hwloop_set_start(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start) +{ + corev_hwloop_set_start(iss, insn, index, start); +} + +static inline void hwloop_set_end(Iss *iss, iss_insn_t *insn, int index, iss_reg_t end) +{ + corev_hwloop_set_end(iss, insn, index, end); +} + +static inline void hwloop_set_count(Iss *iss, iss_insn_t *insn, int index, iss_reg_t count) +{ + corev_hwloop_set_count(iss, insn, index, count); +} + +static inline void hwloop_set_all(Iss *iss, iss_insn_t *insn, int index, iss_reg_t start, iss_reg_t end, iss_reg_t count) +{ + corev_hwloop_set_all(iss, insn, index, start, end, count); +} + +// CV32E40P CoreV2 encodes the hwloop immediate as a word offset (x4); the legacy +// PulpV2 default was a halfword offset (x2). The else-branch keeps x2 for the +// other targets. +/* On iss_v2 this header is only pulled in by the CoreV2 subset, so the + * CV32E40P word-offset encoding applies there unconditionally. A v2 port + * of the legacy halfword encoding must introduce its own gate here. */ +#if defined(CONFIG_GVSOC_ISS_CV32E40P) || defined(CONFIG_GVSOC_ISS_V2) +#define COREV_HWLOOP_IMM_SHIFT 2 +#else +#define COREV_HWLOOP_IMM_SHIFT 1 +#endif + +static inline iss_reg_t lp_starti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - corev_hwloop_set_start(iss, insn, UIM_GET(0), pc + (UIM_GET(1) << 1)); + hwloop_set_start(iss, insn, UIM_GET(0), pc + (UIM_GET(1) << COREV_HWLOOP_IMM_SHIFT)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_endi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t lp_endi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - corev_hwloop_set_end(iss, insn, UIM_GET(0), pc + (UIM_GET(1) << 1)); + hwloop_set_end(iss, insn, UIM_GET(0), pc + (UIM_GET(1) << COREV_HWLOOP_IMM_SHIFT)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_count_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t lp_count_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - corev_hwloop_set_count(iss, insn, UIM_GET(0), REG_GET(0)); + iss->regfile.memcheck_branch_reg(REG_IN(0)); + + hwloop_set_count(iss, insn, UIM_GET(0), REG_GET(0)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_counti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t lp_counti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - corev_hwloop_set_count(iss, insn, UIM_GET(0), UIM_GET(1)); + hwloop_set_count(iss, insn, UIM_GET(0), UIM_GET(1)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_setup_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t lp_setup_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_branch_reg(REG_IN(0)); + int index = UIM_GET(0); iss_reg_t count = REG_GET(0); iss_reg_t start = pc + insn->size; - iss_reg_t end = pc + (UIM_GET(1) << 1); + iss_reg_t end = pc + (UIM_GET(1) << COREV_HWLOOP_IMM_SHIFT); - corev_hwloop_set_all(iss, insn, index, start, end, count); + hwloop_set_all(iss, insn, index, start, end, count); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_setupi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t lp_setupi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { int index = UIM_GET(0); iss_reg_t count = UIM_GET(1); iss_reg_t start = pc + insn->size; - iss_reg_t end = pc + (UIM_GET(2) << 1); + iss_reg_t end = pc + (UIM_GET(2) << COREV_HWLOOP_IMM_SHIFT); - corev_hwloop_set_all(iss, insn, index, start, end, count); + hwloop_set_all(iss, insn, index, start, end, count); return iss_insn_next(iss, insn, pc); } +#endif -static inline iss_reg_t cv_abs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_abs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL1(lib_ABS, REG_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_elw_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t SB_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.store(insn, REG_GET(0) + REG_GET(2), 1, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SB_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(2)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(2)); + iss->lsu.stack_access_check(REG_IN(2), REG_GET(0) + REG_GET(2)); + if (iss->lsu.store_perf(insn, REG_GET(0) + REG_GET(2), 1, REG_IN(1))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SH_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.store(insn, REG_GET(0) + REG_GET(2), 2, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SH_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(2)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(2)); + iss->lsu.stack_access_check(REG_IN(2), REG_GET(0) + REG_GET(2)); + if (iss->lsu.store_perf(insn, REG_GET(0) + REG_GET(2), 2, REG_IN(1))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SW_RR_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->lsu.store(insn, REG_GET(0) + REG_GET(2), 4, REG_IN(1))) + { + // This returns true if the core didn't manage to do the access and is stalled. + return pc; + } + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t SW_RR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // If address register is not valid, we are accessing random location, trigger + // a memcheck fail + iss->regfile.memcheck_access_reg(REG_IN(0)); + iss->regfile.memcheck_access_reg(REG_IN(2)); + + iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + REG_GET(2)); + iss->lsu.stack_access_check(REG_IN(2), REG_GET(0) + REG_GET(2)); + if (iss->lsu.store_perf(insn, REG_GET(0) + REG_GET(2), 4, REG_IN(1))) + { + return pc; + } + return iss_insn_next(iss, insn, pc); +} + + +static inline iss_reg_t p_elw_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_V2 + // No event-load support on this core's LSU: behave as a plain load + // (same fallback as pulp_v2.hpp when CONFIG_GVSOC_ISS_ELW is absent). + iss->regfile.memcheck_access_reg(REG_IN(0)); + if (iss->lsu.load_signed(insn, REG_GET(0) + SIM_GET(0), 4, REG_OUT(0))) + { + return pc; + } +#else + iss->regfile.memcheck_branch_reg(REG_IN(0)); + iss_handle_elw(iss, insn, pc, REG_GET(0) + SIM_GET(0), 4, REG_OUT(0)); +#endif return iss_insn_next(iss, insn, pc); } -#define CV_OP_RS_EXEC(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP_RS_EXEC(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_int16_t_to_int32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sc_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sc_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_int16_t_to_int32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sci_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sci_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_int16_t_to_int32_t, REG_GET(0), SIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_int8_t_to_int32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sc_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sc_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_int8_t_to_int32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sci_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sci_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_int8_t_to_int32_t, REG_GET(0), SIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } -#define CV_OP_RU_EXEC(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP_RU_EXEC(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_uint16_t_to_uint32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sc_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sc_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_uint16_t_to_uint32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sci_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sci_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_uint16_t_to_uint32_t, REG_GET(0), UIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_uint8_t_to_uint32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sc_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sc_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_uint8_t_to_uint32_t, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_sci_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_sci_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_uint8_t_to_uint32_t, REG_GET(0), UIM_GET(0))); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ return iss_insn_next(iss, insn, pc); \ } -#define CV_OP_RS_EXEC2(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP_RS_EXEC2(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_16, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_16, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_16, REG_GET(0), SIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_8, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_8, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_8, REG_GET(0), SIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } -#define CV_OP_RU_EXEC2(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP_RU_EXEC2(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_16, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_16, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_16, REG_GET(0), UIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_8, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_8, REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL2(lib_VEC_##lib_name##_SC_8, REG_GET(0), UIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } -#define CV_OP_RRS_EXEC2(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP_RRS_EXEC2(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_16, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_16, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_16, REG_GET(0), REG_GET(1), SIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_8, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_8, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_8, REG_GET(0), REG_GET(1), SIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } -#define CV_OP_RRU_EXEC2(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP_RRU_EXEC2(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_16, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_16, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_16, REG_GET(0), REG_GET(1), UIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_8, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_8, REG_GET(2), REG_GET(0), REG_GET(1))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); \ REG_SET(0, LIB_CALL3(lib_VEC_##lib_name##_SC_8, REG_GET(0), REG_GET(1), UIM_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } -#define CV_OP1_RS_EXEC(insn_name, lib_name) \ - static inline iss_reg_t cv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ +#define PV_OP1_RS_EXEC(insn_name, lib_name) \ + static inline iss_reg_t pv_##insn_name##_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL1(lib_VEC_##lib_name##_int16_t_to_int32_t, REG_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } \ \ - static inline iss_reg_t cv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ + static inline iss_reg_t pv_##insn_name##_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) \ { \ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); \ REG_SET(0, LIB_CALL1(lib_VEC_##lib_name##_int8_t_to_int32_t, REG_GET(0))); \ return iss_insn_next(iss, insn, pc); \ } -CV_OP_RS_EXEC(add, ADD) +PV_OP_RS_EXEC(add, ADD) -CV_OP_RS_EXEC(sub, SUB) +PV_OP_RS_EXEC(sub, SUB) -CV_OP_RS_EXEC(avg, AVG) +PV_OP_RS_EXEC(avg, AVG) -CV_OP_RU_EXEC(avgu, AVGU) +PV_OP_RU_EXEC(avgu, AVGU) -CV_OP_RS_EXEC(min, MIN) +PV_OP_RS_EXEC(min, MIN) -CV_OP_RU_EXEC(minu, MINU) +PV_OP_RU_EXEC(minu, MINU) -CV_OP_RS_EXEC(max, MAX) +PV_OP_RS_EXEC(max, MAX) -CV_OP_RU_EXEC(maxu, MAXU) +PV_OP_RU_EXEC(maxu, MAXU) -CV_OP_RU_EXEC(srl, SRL) +PV_OP_RU_EXEC(srl, SRL) -CV_OP_RS_EXEC(sra, SRA) +PV_OP_RS_EXEC(sra, SRA) -CV_OP_RU_EXEC(sll, SLL) +PV_OP_RU_EXEC(sll, SLL) -CV_OP_RS_EXEC(or, OR) +PV_OP_RS_EXEC(or, OR) -CV_OP_RS_EXEC(xor, XOR) +PV_OP_RS_EXEC(xor, XOR) -CV_OP_RS_EXEC(and, AND) +PV_OP_RS_EXEC(and, AND) -CV_OP1_RS_EXEC(abs, ABS) +PV_OP1_RS_EXEC(abs, ABS) -static inline iss_reg_t cv_extractr_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_extract_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_EXT_16, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extractr_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_extract_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_EXT_8, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extractur_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_extractu_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_EXTU_16, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extractur_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_extractu_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_EXTU_8, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_insertr_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_insert_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_VEC_INS_16, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_insertr_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_insert_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_VEC_INS_8, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -CV_OP_RS_EXEC2(dotsp, DOTSP) +PV_OP_RS_EXEC2(dotsp, DOTSP) -CV_OP_RU_EXEC2(dotup, DOTUP) +PV_OP_RU_EXEC2(dotup, DOTUP) -CV_OP_RS_EXEC2(dotusp, DOTUSP) +PV_OP_RS_EXEC2(dotusp, DOTUSP) -CV_OP_RRS_EXEC2(sdotsp, SDOTSP) +PV_OP_RRS_EXEC2(sdotsp, SDOTSP) -CV_OP_RRU_EXEC2(sdotup, SDOTUP) +PV_OP_RRU_EXEC2(sdotup, SDOTUP) -CV_OP_RRS_EXEC2(sdotusp, SDOTUSP) +PV_OP_RRS_EXEC2(sdotusp, SDOTUSP) -static inline iss_reg_t cv_shuffle_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shuffle_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLE_16, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shuffle_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shuffle_h_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLE_SCI_16, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shuffle_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shuffle_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLE_8, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shufflei0_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shufflei0_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLEI0_SCI_8, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shufflei1_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shufflei1_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLEI1_SCI_8, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shufflei2_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shufflei2_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLEI2_SCI_8, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shufflei3_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shufflei3_b_sci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); REG_SET(0, LIB_CALL2(lib_VEC_SHUFFLEI3_SCI_8, REG_GET(0), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shuffle2_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shuffle2_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_VEC_SHUFFLE2_16, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_shuffle2_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_shuffle2_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_VEC_SHUFFLE2_8, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_pack_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t pv_pack_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_PACK_SC_16, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t pv_packhi_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL3(lib_VEC_PACKHI_SC_8, REG_GET(0), REG_GET(1), REG_GET(2))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t pv_packlo_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL3(lib_VEC_PACKLO_SC_8, REG_GET(0), REG_GET(1), REG_GET(2))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_pack_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_VEC_PACK_SC_16, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } +static inline iss_reg_t cv_pack_h_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_PACK_SC_H_16, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + static inline iss_reg_t cv_packhi_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_VEC_PACKHI_SC_8, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t cv_packlo_b_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_VEC_PACKLO_SC_8, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -CV_OP_RS_EXEC(cmpeq, CMPEQ) +static inline iss_reg_t cv_add_div2_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_int16_t_to_int32_t_div2, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_add_div4_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_int16_t_to_int32_t_div4, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_add_div8_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_int16_t_to_int32_t_div8, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_sub_div2_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_SUB_int16_t_to_int32_t_div2, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_sub_div4_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_SUB_int16_t_to_int32_t_div4, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_sub_div8_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_SUB_int16_t_to_int32_t_div8, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_subrotmj_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_16_ROTMJ, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_subrotmj_div2_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_16_ROTMJ_DIV2, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_subrotmj_div4_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_16_ROTMJ_DIV4, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_subrotmj_div8_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + REG_SET(0, LIB_CALL2(lib_VEC_ADD_16_ROTMJ_DIV8, REG_GET(0), REG_GET(1))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxconj_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + REG_SET(0, LIB_CALL1(lib_CPLX_CONJ_16, REG_GET(0))); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_r_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_R, REG_GET(0), REG_GET(1), REG_GET(2), 0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_r_div2_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_R, REG_GET(0), REG_GET(1), REG_GET(2), 1)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_r_div4_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_R, REG_GET(0), REG_GET(1), REG_GET(2), 2)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_r_div8_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_R, REG_GET(0), REG_GET(1), REG_GET(2), 3)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_i_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_I, REG_GET(0), REG_GET(1), REG_GET(2), 0)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_i_div2_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_I, REG_GET(0), REG_GET(1), REG_GET(2), 1)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_i_div4_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_I, REG_GET(0), REG_GET(1), REG_GET(2), 2)); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t cv_cplxmul_i_div8_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_CPLXMUL_H_I, REG_GET(0), REG_GET(1), REG_GET(2), 3)); + return iss_insn_next(iss, insn, pc); +} + +PV_OP_RS_EXEC(cmpeq, CMPEQ) -CV_OP_RS_EXEC(cmpne, CMPNE) +PV_OP_RS_EXEC(cmpne, CMPNE) -CV_OP_RS_EXEC(cmpgt, CMPGT) +PV_OP_RS_EXEC(cmpgt, CMPGT) -CV_OP_RS_EXEC(cmpge, CMPGE) +PV_OP_RS_EXEC(cmpge, CMPGE) -CV_OP_RS_EXEC(cmplt, CMPLT) +PV_OP_RS_EXEC(cmplt, CMPLT) -CV_OP_RS_EXEC(cmple, CMPLE) +PV_OP_RS_EXEC(cmple, CMPLE) -CV_OP_RU_EXEC(cmpgtu, CMPGTU) +PV_OP_RU_EXEC(cmpgtu, CMPGTU) -CV_OP_RU_EXEC(cmpgeu, CMPGEU) +PV_OP_RU_EXEC(cmpgeu, CMPGEU) -CV_OP_RU_EXEC(cmpltu, CMPLTU) +PV_OP_RU_EXEC(cmpltu, CMPLTU) -CV_OP_RU_EXEC(cmpleu, CMPLEU) +PV_OP_RU_EXEC(cmpleu, CMPLEU) -static inline iss_reg_t cv_bneimm_exec_common(Iss *iss, iss_insn_t *insn, iss_reg_t pc, int perf) +static inline iss_reg_t p_bneimm_exec_common(Iss *iss, iss_insn_t *insn, iss_reg_t pc, int perf) { + iss->regfile.memcheck_branch_reg(REG_IN(0)); + if ((int32_t)REG_GET(0) != SIM_GET(1)) { #if defined(CONFIG_GVSOC_ISS_V2) @@ -728,18 +2114,20 @@ static inline iss_reg_t cv_bneimm_exec_common(Iss *iss, iss_insn_t *insn, iss_re } } -static inline iss_reg_t cv_bneimm_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_bneimm_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - return cv_bneimm_exec_common(iss, insn, pc, 0); + return p_bneimm_exec_common(iss, insn, pc, 0); } -static inline iss_reg_t cv_bneimm_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_bneimm_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - return cv_bneimm_exec_common(iss, insn, pc, 1); + return p_bneimm_exec_common(iss, insn, pc, 1); } -static inline iss_reg_t cv_beqimm_exec_common(Iss *iss, iss_insn_t *insn, iss_reg_t pc, int perf) +static inline iss_reg_t p_beqimm_exec_common(Iss *iss, iss_insn_t *insn, iss_reg_t pc, int perf) { + iss->regfile.memcheck_branch_reg(REG_IN(0)); + if ((int32_t)REG_GET(0) == SIM_GET(1)) { #if defined(CONFIG_GVSOC_ISS_V2) @@ -759,316 +2147,434 @@ static inline iss_reg_t cv_beqimm_exec_common(Iss *iss, iss_insn_t *insn, iss_re } } -static inline iss_reg_t cv_beqimm_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_beqimm_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - return cv_beqimm_exec_common(iss, insn, pc, 0); + return p_beqimm_exec_common(iss, insn, pc, 0); } -static inline iss_reg_t cv_beqimm_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_beqimm_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - return cv_beqimm_exec_common(iss, insn, pc, 1); + return p_beqimm_exec_common(iss, insn, pc, 1); } -static inline iss_reg_t cv_mac_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mac_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_MAC, REG_GET(2), REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_msu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_msu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_MSU, REG_GET(2), REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mul_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mul_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_MULS, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_muls_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_muls_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_MUL_SL_SL, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulhhs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulhhs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_MUL_SH_SH, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_SL_SL_NR, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulhhsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulhhsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_SH_SH_NR, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulsRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulsNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_SL_SL_NR_R, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulhhsRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulhhsNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_SH_SH_NR_R, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_MUL_ZL_ZL, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulhhu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulhhu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL2(lib_MUL_ZH_ZH, REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_muluN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_muluN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_ZL_ZL_NR, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulhhuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulhhuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_ZH_ZH_NR, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_muluRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_muluNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_ZL_ZL_NR_R, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_mulhhuRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_mulhhuNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_MUL_ZH_ZH_NR_R, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_macs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_macs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_MAC_SL_SL, REG_GET(2), REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_machhs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_machhs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_MAC_SH_SH, REG_GET(2), REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_macsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_macsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL4(lib_MAC_SL_SL_NR, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_machhsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_machhsN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL4(lib_MAC_SH_SH_NR, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_macsRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_macsNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL4(lib_MAC_SL_SL_NR_R, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_machhsRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_machhsNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL4(lib_MAC_SH_SH_NR_R, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_macu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_macu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_MAC_ZL_ZL, REG_GET(2), REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_machhu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_machhu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_MAC_ZH_ZH, REG_GET(2), REG_GET(0), REG_GET(1))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_macuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_macuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL4(lib_MAC_ZL_ZL_NR, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_machhuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_machhuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL4(lib_MAC_ZH_ZH_NR, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_macuRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_macuNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - REG_SET(0, LIB_CALL4(lib_MAC_ZL_ZL_NR_R, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_MAC_ZL_ZL_NR_R_cv32e40p, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_machhuRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_machhuNR_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - REG_SET(0, LIB_CALL4(lib_MAC_ZH_ZH_NR_R, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); + REG_SET(0, LIB_CALL4(lib_MAC_ZH_ZH_NR_R_cv32e40p, REG_GET(2), REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_addN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_addNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_ADD_NR, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_adduN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_adduNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_ADD_NRU, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_addRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_addRNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_ADD_NR_R, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_adduRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_adduRNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_ADD_NR_RU, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_SUB_NR, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subuNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_SUB_NRU, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subRNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_SUB_NR_R, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subuRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subuRNi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); REG_SET(0, LIB_CALL3(lib_SUB_NR_RU, REG_GET(0), REG_GET(1), UIM_GET(0))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_addNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_addN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_ADD_NR, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_adduNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_adduN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_ADD_NRU, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_addRNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_addRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_ADD_NR_R, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_adduRNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_adduRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_ADD_NR_RU, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_SUB_NR, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subuNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subuN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_SUB_NRU, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subRNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_SUB_NR_R, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_subuRNr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_subuRN_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(2)); REG_SET(0, LIB_CALL3(lib_SUB_NR_RU, REG_GET(0), REG_GET(1), REG_GET(2))); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_clip_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_clipi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); int low = (int)-(1 << MAX((int)UIM_GET(0) - 1, 0)); int high = (1 << MAX((int)UIM_GET(0) - 1, 0)) - 1; REG_SET(0, LIB_CALL3(lib_CLIP, REG_GET(0), low, high)); + return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_clipu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_clipui_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - int low = 0; + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); int high = (1 << MAX((int)UIM_GET(0) - 1, 0)) - 1; - REG_SET(0, LIB_CALL3(lib_CLIP, REG_GET(0), low, high)); + REG_SET(0, LIB_CALL2(lib_CLIPU, REG_GET(0), high)); + return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_clipr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_clip_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); int low = -REG_GET(1) - 1; int high = REG_GET(1); REG_SET(0, LIB_CALL3(lib_CLIP, REG_GET(0), low, high)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_clipur_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_clipu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - int low = 0; + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); int high = REG_GET(1); - REG_SET(0, LIB_CALL3(lib_CLIP, REG_GET(0), low, high)); + REG_SET(0, LIB_CALL2(lib_CLIPU, REG_GET(0), high)); + return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_bclr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_bclri_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { int width = UIM_GET(0) + 1; int shift = UIM_GET(1); REG_SET(0, LIB_CALL2(lib_BCLR, REG_GET(0), ((1ULL << width) - 1) << shift)); + iss->regfile.memcheck_set(REG_OUT(0), + iss->regfile.memcheck_get(REG_IN(0)) | ((1ULL << width) - 1) << shift); + return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_bclrr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_bclr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { int width = ((REG_GET(1) >> 5) & 0x1f) + 1; int shift = REG_GET(1) & 0x1f; @@ -1076,74 +2582,81 @@ static inline iss_reg_t cv_bclrr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extract_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_extracti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); int width = UIM_GET(0) + 1; int shift = UIM_GET(1); REG_SET(0, LIB_CALL3(lib_BEXTRACT, REG_GET(0), width, shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extractu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_extractui_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); int width = UIM_GET(0) + 1; int shift = UIM_GET(1); REG_SET(0, LIB_CALL3(lib_BEXTRACTU, REG_GET(0), ((1ULL << width) - 1) << shift, shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extractr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_extract_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); int width = ((REG_GET(1) >> 5) & 0x1f) + 1; int shift = REG_GET(1) & 0x1f; REG_SET(0, LIB_CALL3(lib_BEXTRACT, REG_GET(0), width, shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_extractur_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_extractu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); int width = ((REG_GET(1) >> 5) & 0x1f) + 1; int shift = REG_GET(1) & 0x1f; REG_SET(0, LIB_CALL3(lib_BEXTRACTU, REG_GET(0), ((1ULL << width) - 1) << shift, shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_insert_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_inserti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); int width = UIM_GET(0) + 1; int shift = UIM_GET(1); REG_SET(0, LIB_CALL4(lib_BINSERT, REG_GET(0), REG_GET(1), ((1ULL << width) - 1) << shift, shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_insertr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_insert_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); int width = ((REG_GET(2) >> 5) & 0x1F) + 1; int shift = REG_GET(2) & 0x1F; REG_SET(0, LIB_CALL4(lib_BINSERT, REG_GET(0), REG_GET(1), ((1ULL << width) - 1) << shift, shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_bset_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_bseti_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); int width = UIM_GET(0) + 1; int shift = UIM_GET(1); REG_SET(0, LIB_CALL2(lib_BSET, REG_GET(0), ((1ULL << (width)) - 1) << shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_bsetr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +static inline iss_reg_t p_bset_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(0)); + iss->regfile.memcheck_merge(REG_OUT(0), REG_IN(1)); int width = ((REG_GET(1) >> 5) & 0x1f) + 1; int shift = REG_GET(1) & 0x1f; REG_SET(0, LIB_CALL2(lib_BSET, REG_GET(0), ((1ULL << (width)) - 1) << shift)); return iss_insn_next(iss, insn, pc); } -static inline iss_reg_t cv_bitrev_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) -{ - REG_SET(0, LIB_CALL3(lib_BITREV, REG_GET(0), UIM_GET(0), UIM_GET(1) + 1)); - return iss_insn_next(iss, insn, pc); -} #endif diff --git a/models/cpu/iss/include/isa/rv32c.hpp b/models/cpu/iss/include/isa/rv32c.hpp index c0ad22227..8f916c004 100644 --- a/models/cpu/iss/include/isa/rv32c.hpp +++ b/models/cpu/iss/include/isa/rv32c.hpp @@ -37,13 +37,39 @@ static inline iss_reg_t c_unimp_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) return pc; } +/* Reserved RVC code-points: the ISA table wildcards the immediate/register + * fields, so these encodings decode into the regular handlers, but the RVC + * spec reserves them and the hardware raises illegal instruction (e.g. + * cv32e40p_compressed_decoder.sv). Mirrors iss_exec_insn_illegal. + * The checks below are gated by CONFIG_GVSOC_ISS_RVC_STRICT (opt-in per + * core, e.g. cv32e40p_v2.py): the historical permissive decoding - and + * the golden traces built on it - stays the default for every other + * core. */ +static inline iss_reg_t c_reserved_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->decode.trace.msg("Executing illegal instruction\n"); + iss->exception.raise(pc, ISS_EXCEPT_ILLEGAL); +#ifdef CONFIG_GVSOC_ISS_V2 + iss->exec.insn_stall(); +#endif + return pc; +} + static inline iss_reg_t c_addi4spn_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x1FE0) == 0) /* nzuimm == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif return addi_exec(iss, insn, pc); } static inline iss_reg_t c_addi4spn_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x1FE0) == 0) /* nzuimm == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif iss->timing.event_rvc_account(1); return addi_exec(iss, insn, pc); } @@ -127,11 +153,19 @@ static inline iss_reg_t c_li_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t c_addi16sp_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x107C) == 0) /* nzimm == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif return addi_exec(iss, insn, pc); } static inline iss_reg_t c_addi16sp_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x107C) == 0) /* nzimm == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif iss->timing.event_rvc_account(1); return addi_exec(iss, insn, pc); } @@ -149,11 +183,19 @@ static inline iss_reg_t c_jalr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t c_lui_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x107C) == 0) /* imm == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif return lui_exec(iss, insn, pc); } static inline iss_reg_t c_lui_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x107C) == 0) /* imm == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif iss->timing.event_rvc_account(1); return lui_exec(iss, insn, pc); } @@ -281,22 +323,38 @@ static inline iss_reg_t c_slli_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t c_lwsp_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x0F80) == 0) /* rd == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif return lw_exec_fast(iss, insn, pc); } static inline iss_reg_t c_lwsp_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x0F80) == 0) /* rd == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif iss->timing.event_rvc_account(1); return lw_exec(iss, insn, pc); } static inline iss_reg_t c_jr_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x0F80) == 0) /* rs1 == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif return jalr_exec_fast(iss, insn, pc); } static inline iss_reg_t c_jr_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + if ((insn->opcode & 0x0F80) == 0) /* rs1 == 0: reserved */ + return c_reserved_exec(iss, insn, pc); +#endif iss->timing.event_rvc_account(1); return jalr_exec(iss, insn, pc); } @@ -327,22 +385,54 @@ static inline iss_reg_t c_ebreak_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { iss->timing.event_rvc_account(1); - if ((iss->csr.dcsr >> 15) & 1) + // Same semantics as the 32-bit ebreak (rv32i.hpp ebreak_exec), minus the + // semihosting probe which is specified on the uncompressed form only. + if (iss->exec.debug_mode) { - // iss->dbgunit.set_halt_mode(true, 1); + // Back to the park loop: an ebreak in debug mode re-enters the + // debug ROM instead of raising an exception. + return iss->irq.debug_handler; } - else +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // dcsr.ebreakm=1 in M-mode: enter debug (RISC-V Debug Spec). + if (iss->csr.ebreak_m_mode_enters_debug()) { - iss->exception.raise(pc, ISS_EXCEPT_BREAKPOINT); + iss->dbgunit.set_halt_mode(true, HALT_CAUSE_EBREAK); + return pc; } - return iss_insn_next(iss, insn, pc); +#endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P_V2 + // dcsr.ebreakm=1 in M-mode: enter debug (RISC-V Debug Spec). + // Arms the request (cause=1); check() performs the entry at the next + // dispatch boundary - see rv32i.hpp ebreak_exec. + if (iss->csr.ebreak_m_mode_enters_debug()) + { + iss->irq.ebreak_enter_debug(); + return pc; + } +#endif + iss->exception.raise(pc, ISS_EXCEPT_BREAKPOINT); + return pc; } static inline iss_reg_t c_sbreak_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#ifdef CONFIG_GVSOC_ISS_RVC_STRICT + /* 0x8002 is c.jr with rs1=0: an RVC-reserved code-point. The legacy + * PULP c.sbreak (RI5CY era) repurposed this encoding as a nop-like + * debug break, so the ISA table decodes it here instead of c_jr_exec + * (whose rs1==0 strict check can therefore never fire). Cores that + * opt into strict RVC decoding (e.g. CV32E40P, whose compressed + * decoder raises illegal instruction on it) must trap HERE, with + * mepc = the faulting pc: falling through to the nop would retire a + * phantom instruction and take the illegal one insn later (off-by-2 + * mepc against the RTL). */ + return c_reserved_exec(iss, insn, pc); +#else iss->timing.event_rvc_account(1); // iss->dbgunit.set_halt_mode(true, 3); return iss_insn_next(iss, insn, pc); +#endif } #endif diff --git a/models/cpu/iss/include/isa/rv32i.hpp b/models/cpu/iss/include/isa/rv32i.hpp index 9b1d83424..0da9eb037 100644 --- a/models/cpu/iss/include/isa/rv32i.hpp +++ b/models/cpu/iss/include/isa/rv32i.hpp @@ -784,6 +784,25 @@ static inline iss_reg_t ebreak_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { return iss->irq.debug_handler; } +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // dcsr.ebreakm=1 in M-mode: enter debug (RISC-V Debug Spec). + else if (iss->csr.ebreak_m_mode_enters_debug()) + { + iss->dbgunit.set_halt_mode(true, HALT_CAUSE_EBREAK); + return pc; + } +#endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P_V2 + // dcsr.ebreakm=1 in M-mode: enter debug (RISC-V Debug Spec). + // Arms the request (cause=1); Cv32e40pIrq::check() performs the entry + // at the next dispatch boundary, so the ebreak never retires and + // mcause/mepc stay untouched, as in the hardware. + else if (iss->csr.ebreak_m_mode_enters_debug()) + { + iss->irq.ebreak_enter_debug(); + return pc; + } +#endif else { iss->exception.raise(pc, ISS_EXCEPT_BREAKPOINT); diff --git a/models/cpu/iss/include/isa/rvf.hpp b/models/cpu/iss/include/isa/rvf.hpp index e7fa82626..5eabcbd21 100644 --- a/models/cpu/iss/include/isa/rvf.hpp +++ b/models/cpu/iss/include/isa/rvf.hpp @@ -35,8 +35,24 @@ #include "cpu/iss/include/isa_lib/float.h" #include "cpu/iss/include/isa/rvd.hpp" +// On cores that track mstatus.FS (CONFIG_GVSOC_ISS_FP_STATE_DIRTY) every +// completed FPU op dirties FS: the RTL raises fflags_we on apu_valid even +// when no flag ends up set, so GPR-writing FP ops (compare, convert, +// classify, fmv.x) dirty it too. FP register write-backs already get this +// from the FREG*_SET macros; call this after the write-back so a raise +// (reserved rounding mode) has already set has_exception and is skipped. +#if defined(CONFIG_GVSOC_ISS_FP_STATE_DIRTY) +#define FP_EXEC_DIRTY() (iss->csr.fp_state_dirty()) +#else +#define FP_EXEC_DIRTY() ((void)0) +#endif + static inline iss_reg_t flw_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { +#if defined(CONFIG_GVSOC_ISS_FP_STATE_DIRTY) + // Opt-in: FP loads dirty mstatus.FS (see iss_v2 isa_lib/macros.h). + iss->csr.fp_state_dirty(); +#endif if (iss->lsu.load_float(insn, REG_GET(0) + SIM_GET(0), 4, REG_OUT(0))) { return pc; @@ -47,6 +63,9 @@ static inline iss_reg_t flw_exec_fast(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t flw_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { iss->lsu.stack_access_check(REG_IN(0), REG_GET(0) + SIM_GET(0)); +#if defined(CONFIG_GVSOC_ISS_FP_STATE_DIRTY) + iss->csr.fp_state_dirty(); +#endif if (iss->lsu.load_float_perf(insn, REG_GET(0) + SIM_GET(0), 4, REG_OUT(0))) { return pc; @@ -174,18 +193,21 @@ static inline iss_reg_t fmax_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t fcvt_w_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_cvt_w_ff_round, FREG32_GET(0), 8, 23, UIM_GET(0))); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t fcvt_wu_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_cvt_wu_ff_round, FREG32_GET(0), 8, 23, UIM_GET(0))); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t fmv_x_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL1(lib_flexfloat_fmv_x_ff, FREG32_GET(0), 8, 23)); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } @@ -198,24 +220,28 @@ static inline iss_reg_t fmv_s_x_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t feq_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_eq, FREG32_GET(0), FREG32_GET(1), 8, 23)); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t flt_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_lt, FREG32_GET(0), FREG32_GET(1), 8, 23)); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t fle_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_le, FREG32_GET(0), FREG32_GET(1), 8, 23)); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t fclass_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL1(lib_flexfloat_class, FREG32_GET(0), 8, 23)); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } @@ -237,12 +263,14 @@ static inline iss_reg_t fcvt_s_wu_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t fcvt_l_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_cvt_l_ff_round, FREG32_GET(0), 8, 23, UIM_GET(0))); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } static inline iss_reg_t fcvt_lu_s_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { REG_SET(0, LIB_FF_CALL2(lib_flexfloat_cvt_lu_ff_round, FREG32_GET(0), 8, 23, UIM_GET(0))); + FP_EXEC_DIRTY(); return iss_insn_next(iss, insn, pc); } diff --git a/models/cpu/iss/include/isa_lib/int.h b/models/cpu/iss/include/isa_lib/int.h index b23aeae32..cf1822ab0 100644 --- a/models/cpu/iss/include/isa_lib/int.h +++ b/models/cpu/iss/include/isa_lib/int.h @@ -176,33 +176,36 @@ static inline unsigned int lib_MAC_SH_SH_NR(Iss *s, unsigned int a, unsigned int static inline unsigned int lib_MAC_ZL_ZL_NR(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) { return ((uint32_t)(a + ZL(b) * ZL(c))) >> shift; } static inline unsigned int lib_MAC_ZH_ZH_NR(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) { return ((uint32_t)(a + ZH(b) * ZH(c))) >> shift; } +/* The rounding constant takes part in the 32-bit wrap of the accumulate sum + * (RI5CY-family mult datapath); adding it after the wrap flips the result + * sign when a + product + round crosses the 32-bit boundary. */ static inline unsigned int lib_MAC_SL_SL_NR_R(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) { - int32_t result = (int32_t)(a + SL(b) * SL(c)); + uint32_t result = a + SL(b) * SL(c); if (shift > 0) - result = (result + (1ULL << (shift - 1))) >> shift; - return result; + result += 1u << (shift - 1); + return ((int32_t)result) >> shift; } static inline unsigned int lib_MAC_SH_SH_NR_R(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) { - int32_t result = (int32_t)(a + SH(b) * SH(c)); + uint32_t result = a + SH(b) * SH(c); if (shift > 0) - result = (result + (1ULL << (shift - 1))) >> shift; - return result; + result += 1u << (shift - 1); + return ((int32_t)result) >> shift; } static inline unsigned int lib_MAC_ZL_ZL_NR_R(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) { - uint32_t result = (uint32_t)(a + ZL(b) * ZL(c)); + uint32_t result = a + (uint32_t)ZL(b) * ZL(c); if (shift > 0) - result = (result + (1ULL << (shift - 1))) >> shift; - return result; + result += 1u << (shift - 1); + return result >> shift; } static inline unsigned int lib_MAC_ZH_ZH_NR_R(Iss *s, unsigned int a, unsigned int b, unsigned int c, unsigned int shift) { - uint32_t result = (uint32_t)(a + ZH(b) * ZH(c)); + uint32_t result = a + (uint32_t)ZH(b) * ZH(c); if (shift > 0) - result = (result + (1ULL << (shift - 1))) >> shift; - return result; + result += 1u << (shift - 1); + return result >> shift; } static inline unsigned int lib_MSU_SL_SL(Iss *s, unsigned int a, unsigned int b, unsigned int c) { return a - SL(b) * SL(c); } @@ -1496,6 +1499,13 @@ static inline void clear_fflags(Iss *iss, unsigned long int fflags) // updates the fflags from fenv exceptions static inline void update_fflags_fenv(Iss *iss) { +#if defined(CONFIG_GVSOC_ISS_CV32E40P) || defined(CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS) + // A trapped FP instruction (reserved rounding mode, see setFFRoundingMode) + // must not update fflags: the RTL kills every side effect of an + // instruction that raised illegal-instruction. + if (iss->exec.has_exception) + return; +#endif int ex = fetestexcept(FE_ALL_EXCEPT); int flags = !!(ex & FE_INEXACT) | !!(ex & FE_UNDERFLOW) << 1 | @@ -1511,15 +1521,20 @@ static inline int32_t double_to_int(Iss *s, double dbl_i) { double dbl = nearbyint(dbl_i); - if (dbl != dbl_i) - { - set_fflags(s, 1ULL << 0); - } - if (dbl < 2.0 * (INT32_MAX / 2 + 1)) { // NO OVERFLOW if (ceil(dbl) >= INT32_MIN) // NO UNDERFLOW + { + // Inexact belongs to a VALID conversion only: an invalid float to + // integer conversion (NaN or out of range) raises NV alone + // (IEEE 754 7.2, RISC-V unpriv F). The test used to fire on NaN + // too, since NaN != NaN holds. + if (dbl != dbl_i) + { + set_fflags(s, 1ULL << 0); + } return (int32_t)dbl; + } else // UNDERFLOW { set_fflags(s, 1ULL << 4); @@ -1538,15 +1553,20 @@ static inline uint32_t double_to_uint(Iss *s, double dbl_i) { double dbl = nearbyint(dbl_i); - if (dbl != dbl_i) - { - set_fflags(s, 1ULL << 0); - } - if (dbl < 2.0 * (UINT32_MAX / 2 + 1)) { // NO OVERFLOW if (ceil(dbl) >= 0) // NO UNDERFLOW + { + // Inexact belongs to a VALID conversion only: an invalid float to + // integer conversion (NaN or out of range) raises NV alone + // (IEEE 754 7.2, RISC-V unpriv F). The test used to fire on NaN + // too, since NaN != NaN holds. + if (dbl != dbl_i) + { + set_fflags(s, 1ULL << 0); + } return (uint32_t)dbl; + } else // UNDERFLOW { set_fflags(s, 1ULL << 4); @@ -1565,15 +1585,20 @@ static inline int64_t double_to_long(Iss *s, double dbl_i) { double dbl = nearbyint(dbl_i); - if (dbl != dbl_i) - { - set_fflags(s, 1ULL << 0); - } - if (dbl < 2.0 * (INT64_MAX / 2 + 1)) { // NO OVERFLOW if (ceil(dbl) >= INT64_MIN) // NO UNDERFLOW + { + // Inexact belongs to a VALID conversion only: an invalid float to + // integer conversion (NaN or out of range) raises NV alone + // (IEEE 754 7.2, RISC-V unpriv F). The test used to fire on NaN + // too, since NaN != NaN holds. + if (dbl != dbl_i) + { + set_fflags(s, 1ULL << 0); + } return (int64_t)dbl; + } else // UNDERFLOW { set_fflags(s, 1ULL << 4); @@ -1592,15 +1617,20 @@ static inline uint64_t double_to_ulong(Iss *s, double dbl_i) { double dbl = nearbyint(dbl_i); - if (dbl != dbl_i) - { - set_fflags(s, 1ULL << 0); - } - if (dbl < 2.0 * (UINT64_MAX / 2 + 1)) { // NO OVERFLOW if (ceil(dbl) >= 0) // NO UNDERFLOW + { + // Inexact belongs to a VALID conversion only: an invalid float to + // integer conversion (NaN or out of range) raises NV alone + // (IEEE 754 7.2, RISC-V unpriv F). The test used to fire on NaN + // too, since NaN != NaN holds. + if (dbl != dbl_i) + { + set_fflags(s, 1ULL << 0); + } return (uint64_t)dbl; + } else // UNDERFLOW { set_fflags(s, 1ULL << 4); @@ -1719,9 +1749,25 @@ static inline unsigned long int setFFRoundingMode(Iss *s, unsigned long int mode fesetround(FE_UPWARD); break; case 4: - printf("Unimplemented roudning mode nearest ties to max magnitude"); - exit(-1); + // RMM has no fenv equivalent: nearest plus the flexfloat ties-away flag. + fesetround(FE_TONEAREST); + flexfloat_rmm = 1; + break; +#if defined(CONFIG_GVSOC_ISS_CV32E40P) || defined(CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS) + case 5: + case 6: + // Reserved static rounding modes: illegal-instruction on the RTL + // (RISC-V F spec, rm 101/110 reserved). The FP op still runs after + // the raise; its writeback and fflags update are suppressed by the + // has_exception guards (macros.h FREG_SET, update_fflags_fenv). + s->exception.raise(s->exec.current_insn, ISS_EXCEPT_ILLEGAL); +#ifdef CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS + // The iss_v2 macros have no write-back guard; the regfile + // personality drops the pending write instead. + s->regfile.wb_suppress_arm(); +#endif break; +#endif case 7: { switch (s->csr.fcsr.frm) @@ -1739,9 +1785,21 @@ static inline unsigned long int setFFRoundingMode(Iss *s, unsigned long int mode fesetround(FE_UPWARD); break; case 4: - printf("Unimplemented roudning mode nearest ties to max magnitude"); - exit(-1); + fesetround(FE_TONEAREST); + flexfloat_rmm = 1; break; +#if defined(CONFIG_GVSOC_ISS_CV32E40P) || defined(CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS) + case 5: + case 6: + case 7: + // Dynamic rounding with a reserved frm value: illegal-instruction + // on the RTL (frm 101/110/111 reserved on use). + s->exception.raise(s->exec.current_insn, ISS_EXCEPT_ILLEGAL); +#ifdef CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS + s->regfile.wb_suppress_arm(); +#endif + break; +#endif } } } @@ -1750,6 +1808,7 @@ static inline unsigned long int setFFRoundingMode(Iss *s, unsigned long int mode static inline void restoreFFRoundingMode(unsigned long int mode) { + flexfloat_rmm = 0; fesetround(mode); } @@ -1902,6 +1961,13 @@ static inline unsigned long int lib_flexfloat_min(Iss *s, unsigned long int a, u #ifdef OLD FF_EXEC_2(s, ff_min, a, b, e, m) #else + // IEEE 754-2019 minimumNumber/maximumNumber (the semantics RISC-V gives + // fmin/fmax) signal invalid on a signaling NaN input; the max twin below + // already does it. + if (IsNan(a, e, m) == 2 || IsNan(b, e, m) == 2) + { + set_fflags(s, 1ULL << 4); + } int Nan_a = IsNan(a, e, m); int Nan_b = IsNan(b, e, m); unsigned long int Nan_Q = (((1ULL << e) - 1) << m) | ((unsigned long int)1ULL << (m - 1)); @@ -1960,39 +2026,92 @@ static inline int64_t lib_flexfloat_cvt_w_ff_round(Iss *s, unsigned long int a, unsigned long int new_round = round == 4 ? 2 : round; old = setFFRoundingMode(s, new_round); FF_INIT_1(a, e, m) - if (round == 4) + if (round == 4 && ff_a.value == ff_a.value /* !NaN */) { - if (ff_a.value < 0) + /* RMM(x) = trunc(|x|+0.5) with the sign re-applied (host fenv is in + * RTZ here). The flags must come from the ORIGINAL operand, not the + * nudged one: feeding x+0.5 to the generic converter raised a + * spurious NX for every exact-integer input (x+0.5 truncates) and + * LOST the NX of every half-tie (x+0.5 lands on an integer). On + * saturation NV is raised alone (IEEE 754 7.2 / RISC-V F 11.7); a + * NaN input keeps the generic path below, which raises NV alone. */ + neg = ff_a.value < 0; + double mag = neg ? -ff_a.value : ff_a.value; + double t = trunc(mag + 0.5); /* round-half-away magnitude; exact: + * above 2^52 the +0.5 is absorbed and + * mag is already integral */ + double lim = neg ? 2147483648.0 : 2147483647.0; + int32_t result_rmm; + if (t > lim) + { + set_fflags(s, 1ULL << 4); /* NV alone on out-of-range */ + result_rmm = neg ? INT32_MIN : INT32_MAX; + } + else { - ff_a.value = -ff_a.value; - neg = true; + double rounded = neg ? -t : t; + result_rmm = (int32_t)rounded; + if (rounded != ff_a.value) + set_fflags(s, 1ULL << 0); /* NX iff the rounding moved x */ } - ff_a.value += 0.5f; + restoreFFRoundingMode(old); + return iss_get_signed_value(result_rmm, 32); } int32_t result_int = double_to_int(s, ff_a.value); - if (neg) - { - result_int = -result_int; - } - restoreFFRoundingMode(new_round); + restoreFFRoundingMode(old); return iss_get_signed_value(result_int, 32); } static inline int64_t lib_flexfloat_cvt_wu_ff_round(Iss *s, unsigned long int a, uint8_t e, uint8_t m, unsigned long int round) { int old; - bool neg = false; unsigned long int new_round = round == 4 ? 2 : round; old = setFFRoundingMode(s, new_round); FF_INIT_1(a, e, m) - if (round == 4) + if (round == 4 && ff_a.value == ff_a.value /* !NaN */) { - if (ff_a.value < 0) + /* Same flag discipline as the signed sibling: RMM via trunc(x+0.5) + * on the magnitude, NX decided against the ORIGINAL operand (the + * nudge corrupted it both ways), NV alone on out-of-range. A + * negative operand rounds half-away from zero DOWN: anything that + * rounds below zero is out of range for the unsigned destination + * (NV, result 0) - the old path fed it unrounded to the RTZ + * converter, so RMM(-0.5), an out-of-range -1, came back 0 with a + * mere NX. NaN keeps the generic path (NV alone, all-ones). */ + double v = ff_a.value; + int32_t result_rmm; + if (v < 0) { - ff_a.value = -ff_a.value; - neg = true; + double t = trunc(-v + 0.5); + if (t > 0) + { + set_fflags(s, 1ULL << 4); /* rounds to < 0: out of range */ + result_rmm = 0; + } + else + { + result_rmm = 0; /* RMM(v) == 0 exactly */ + if (v != 0.0) + set_fflags(s, 1ULL << 0); + } + } + else + { + double t = trunc(v + 0.5); + if (t > 4294967295.0) + { + set_fflags(s, 1ULL << 4); + result_rmm = (int32_t)UINT32_MAX; + } + else + { + result_rmm = (int32_t)(uint32_t)t; + if (t != v) + set_fflags(s, 1ULL << 0); + } } - ff_a.value += 0.5f; + restoreFFRoundingMode(old); + return (int64_t)result_rmm; } int32_t result_int = double_to_uint(s, ff_a.value); restoreFFRoundingMode(old); @@ -2003,7 +2122,13 @@ static inline long int lib_flexfloat_cvt_ff_w_round(Iss *s, int64_t a, uint8_t e { int old = setFFRoundingMode(s, round); flexfloat_t ff_a; + // An integer to float conversion is inexact when the integer does not fit + // in the mantissa: flexfloat_sanitize already raises FE_INEXACT in the host + // fenv, it was simply never collected. Same pattern as the vector twin + // lib_flexfloat_cvt_ff_x_round below. + feclearexcept(FE_ALL_EXCEPT); ff_init_int(&ff_a, a & 0xffffffff, (flexfloat_desc_t){e, m}); + update_fflags_fenv(s); restoreFFRoundingMode(old); return flexfloat_get_bits(&ff_a); } @@ -2012,7 +2137,11 @@ static inline unsigned long int lib_flexfloat_cvt_ff_wu_round(Iss *s, int64_t a, { int old = setFFRoundingMode(s, round); flexfloat_t ff_a; + // Inexact when the integer does not fit in the mantissa; see + // lib_flexfloat_cvt_ff_w_round above. + feclearexcept(FE_ALL_EXCEPT); ff_init_long(&ff_a, (uint32_t)a & 0xffffffff, (flexfloat_desc_t){e, m}); + update_fflags_fenv(s); restoreFFRoundingMode(old); return flexfloat_get_bits(&ff_a); } @@ -2039,7 +2168,11 @@ static inline long int lib_flexfloat_cvt_ff_l_round(Iss *s, int64_t a, uint8_t e { int old = setFFRoundingMode(s, round); flexfloat_t ff_a; + // Inexact when the integer does not fit in the mantissa; see + // lib_flexfloat_cvt_ff_w_round above. + feclearexcept(FE_ALL_EXCEPT); ff_init_long(&ff_a, a, (flexfloat_desc_t){e, m}); + update_fflags_fenv(s); restoreFFRoundingMode(old); return flexfloat_get_bits(&ff_a); } @@ -2048,7 +2181,11 @@ static inline unsigned long int lib_flexfloat_cvt_ff_lu_round(Iss *s, uint64_t a { int old = setFFRoundingMode(s, round); flexfloat_t ff_a; + // Inexact when the integer does not fit in the mantissa; see + // lib_flexfloat_cvt_ff_w_round above. + feclearexcept(FE_ALL_EXCEPT); ff_init_long_long_unsigned(&ff_a, a, (flexfloat_desc_t){e, m}); + update_fflags_fenv(s); restoreFFRoundingMode(old); return flexfloat_get_bits(&ff_a); } @@ -2057,7 +2194,13 @@ static inline long int lib_flexfloat_cvt_ff_ff_round(Iss *s, unsigned long int a { int old = setFFRoundingMode(s, round); FF_INIT_1(a, es, ms) + // A narrowing format conversion rounds and can overflow/underflow: + // flexfloat_sanitize raises the host fenv flags inside ff_cast, they + // were simply never collected here - every sibling conversion in this + // file collects them (same pattern as lib_flexfloat_cvt_ff_w_round). + feclearexcept(FE_ALL_EXCEPT); ff_cast(&ff_res, &ff_a, (flexfloat_desc_t){ed, md}); + update_fflags_fenv(s); restoreFFRoundingMode(old); return flexfloat_get_bits(&ff_res); } diff --git a/models/cpu/iss/include/isa_lib/macros.h b/models/cpu/iss/include/isa_lib/macros.h index 8468d5993..aa97f687d 100644 --- a/models/cpu/iss/include/isa_lib/macros.h +++ b/models/cpu/iss/include/isa_lib/macros.h @@ -73,4 +73,47 @@ #define FREG_SET(reg,val) (iss->regfile.set_freg(insn->out_regs[reg], val)) #endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P +// CV32E40P RTL: a trapped instruction has no architectural side effects, and +// an FP regfile write (fregs_we_i) forces mstatus.FS=Dirty +// (cv32e40p_cs_registers.sv). Redefine the write-back macros accordingly: any +// destination write -- integer or FP -- is killed when the instruction raised +// an exception mid-execution (reserved FP rounding modes raise +// illegal-instruction inside the value expression, see setFFRoundingMode; +// fcvt.w.s and friends land in an integer rd, so guarding only the FP macros +// left that path writing a trapped result), and FP writes additionally +// promote FS. Kept in this isolated block (not inline above) so the shared +// definitions preprocess back to upstream byte-for-byte without this define. +// The write must target the same bank as the definitions above: the integer +// regfile on ISS_SINGLE_REGFILE (ZFINX) builds, the FP regfile otherwise +// (FPU=1 ZFINX=0 builds have a separate FP bank; routing these writes to +// the integer bank there corrupts the integer registers). +// The value expression MUST be evaluated before the guard: the raise happens +// inside it, so testing has_exception first would always see the pre-raise +// state. +#undef REG_SET +#define REG_SET(reg,val) \ + do { iss_reg_t int_wb_val_ = (val); \ + if (!iss->exec.has_exception) { \ + iss->regfile.set_reg(insn->out_regs[reg], int_wb_val_); \ + } } while (0) +#undef FREG_SET +#undef FREG32_SET +#ifdef ISS_SINGLE_REGFILE +#define FREG_SET(reg,val) \ + do { iss_reg_t fp_wb_val_ = (val); \ + if (!iss->exec.has_exception) { \ + REG_SET(reg, fp_wb_val_); \ + iss->csr.fp_state_dirty(); \ + } } while (0) +#else +#define FREG_SET(reg,val) \ + do { iss_freg_t fp_wb_val_ = (val); \ + if (!iss->exec.has_exception) { \ + iss->regfile.set_freg(insn->out_regs[reg], fp_wb_val_); \ + iss->csr.fp_state_dirty(); \ + } } while (0) +#endif +#define FREG32_SET(reg,val) FREG_SET(reg, val) +#endif #endif diff --git a/models/cpu/iss/isa_gen/isa_cv32e40pv2.py b/models/cpu/iss/isa_gen/isa_cv32e40pv2.py new file mode 100644 index 000000000..2f210bce1 --- /dev/null +++ b/models/cpu/iss/isa_gen/isa_cv32e40pv2.py @@ -0,0 +1,467 @@ +# +# Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and +# University of Bologna +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +from cpu.iss.isa_gen.isa_gen import * +from cpu.iss.isa_gen.isa_riscv_gen import * +from cpu.iss.isa_gen.isa_corev import Format_SB2, Format_LPOST, Format_LRPOST, Format_LR, Format_SPOST, Format_SRPOST, Format_SR, Format_HL0, Format_HL1, Format_RRRR, Format_RRRR2, Format_RRRU2, Format_RRRRU, Format_R1, Format_I1U, Format_I4U, Format_I5U, Format_I5U2, Format_BITREV + +# Local copies of the SIMD scalar-replication formats from isa_pulpv2.py: they are +# not exported by isa_corev, so the SIMD decode table below needs its own copy. +Format_RRU = [ OutReg(0, Range(7, 5)), + InReg (0, Range(15, 5)), + UnsignedImm(0, Ranges([[25, 1, 0], [20, 5, 1]])), +] +Format_RRU2 = [ OutReg(0, Range(7, 5)), + InReg (0, Range(15, 5)), + UnsignedImm(0, Ranges([[25, 1, 0], [20, 5, 1]])), +] +Format_RRS = [ OutReg(0, Range(7, 5)), + InReg (0, Range(15, 5)), + SignedImm(0, Ranges([[25, 1, 0], [20, 5, 1]])), +] +# Accumulate formats (insert / sdotsp.sci / sdotusp.sci) from isa_pulpv2.py. RRRU +# also lives in isa_corev but is not exported, so a local copy is needed here. +Format_RRRS = [ OutReg(0, Range(7, 5)), + InReg (0, Range(7, 5)), + InReg (1, Range(15, 5)), + SignedImm(0, Ranges([[25, 1, 0], [20, 5, 1]])), +] +Format_RRRU = [ OutReg(0, Range(7, 5)), + InReg (0, Range(7, 5)), + InReg (1, Range(15, 5)), + UnsignedImm(0, Ranges([[25, 1, 0], [20, 5, 1]])), +] + +# CORE-V v2 uses standard RISC-V custom opcodes: +# Custom-0: 0x0B (0001011) - Branching & Load Immediate Post-Inc +# Custom-1: 0x2B (0101011) - Store Immediate Post-Inc & Register Load/Store +# Custom-2: 0x5B (1011011) - Multiplication / MAC / HWLoop + +class CoreV2(IsaSubset): + + # elw mirrors the CV32E40P COREV_CLUSTER parameter (RTL default 0): cv.elw + # raises illegal-instruction unless the core is built for a cluster. + def __init__(self, elw=False): + instrs = [] + + # --- Custom-0 (0x0B): Branching and Load Post-Increment --- + # Note: names must match C++ handler names (case sensitive) + instrs += [ + Instr('p.beqimm', Format_SB2, '------- ----- ----- 110 ----- 0001011', fast_handler=True, decode='bxx_decode', L='cv.beqimm'), + Instr('p.bneimm', Format_SB2, '------- ----- ----- 111 ----- 0001011', fast_handler=True, decode='bxx_decode', L='cv.bneimm'), + + Instr('LB_POSTINC', Format_LPOST, '------- ----- ----- 000 ----- 0001011', L='cv.lb' , fast_handler=True, tags=["load"]), + Instr('LBU_POSTINC', Format_LPOST, '------- ----- ----- 100 ----- 0001011', L='cv.lbu', fast_handler=True, tags=["load"]), + Instr('LH_POSTINC', Format_LPOST, '------- ----- ----- 001 ----- 0001011', L='cv.lh' , fast_handler=True, tags=["load"]), + Instr('LHU_POSTINC', Format_LPOST, '------- ----- ----- 101 ----- 0001011', L='cv.lhu', fast_handler=True, tags=["load"]), + Instr('LW_POSTINC', Format_LPOST, '------- ----- ----- 010 ----- 0001011', L='cv.lw' , fast_handler=True, tags=["load"]), + ] + + if elw: + instrs += [ + Instr('p.elw', Format_L, '------- ----- ----- 011 ----- 0001011', L='cv.elw', tags=["load"]), + ] + + # --- Custom-1 (0x2B): Store Post-Increment and Register Indexed Load/Store --- + instrs += [ + Instr('SB_POSTINC', Format_SPOST, '------- ----- ----- 000 ----- 0101011', L='cv.sb' , fast_handler=True), + Instr('SH_POSTINC', Format_SPOST, '------- ----- ----- 001 ----- 0101011', L='cv.sh' , fast_handler=True), + Instr('SW_POSTINC', Format_SPOST, '------- ----- ----- 010 ----- 0101011', L='cv.sw' , fast_handler=True), + + Instr('LB_RR_POSTINC', Format_LRPOST, '0000000 ----- ----- 011 ----- 0101011', L='cv.lb' , fast_handler=True, tags=["load"]), + Instr('LBU_RR_POSTINC', Format_LRPOST, '0001000 ----- ----- 011 ----- 0101011', L='cv.lbu', fast_handler=True, tags=["load"]), + Instr('LH_RR_POSTINC', Format_LRPOST, '0000001 ----- ----- 011 ----- 0101011', L='cv.lh' , fast_handler=True, tags=["load"]), + Instr('LHU_RR_POSTINC', Format_LRPOST, '0001001 ----- ----- 011 ----- 0101011', L='cv.lhu', fast_handler=True, tags=["load"]), + Instr('LW_RR_POSTINC', Format_LRPOST, '0000010 ----- ----- 011 ----- 0101011', L='cv.lw' , fast_handler=True, tags=["load"]), + + Instr('LB_RR', Format_LR, '0000100 ----- ----- 011 ----- 0101011', L='cv.lb' , fast_handler=True, tags=["load"]), + Instr('LBU_RR', Format_LR, '0001100 ----- ----- 011 ----- 0101011', L='cv.lbu', fast_handler=True, tags=["load"]), + Instr('LH_RR', Format_LR, '0000101 ----- ----- 011 ----- 0101011', L='cv.lh' , fast_handler=True, tags=["load"]), + Instr('LHU_RR', Format_LR, '0001101 ----- ----- 011 ----- 0101011', L='cv.lhu', fast_handler=True, tags=["load"]), + Instr('LW_RR', Format_LR, '0000110 ----- ----- 011 ----- 0101011', L='cv.lw' , fast_handler=True, tags=["load"]), + + Instr('SB_RR_POSTINC', Format_SRPOST, '0010000 ----- ----- 011 ----- 0101011', L='cv.sb' , fast_handler=True), + Instr('SH_RR_POSTINC', Format_SRPOST, '0010001 ----- ----- 011 ----- 0101011', L='cv.sh' , fast_handler=True), + Instr('SW_RR_POSTINC', Format_SRPOST, '0010010 ----- ----- 011 ----- 0101011', L='cv.sw' , fast_handler=True), + + Instr('SB_RR', Format_SR, '0010100 ----- ----- 011 ----- 0101011', L='cv.sb', fast_handler=True), + Instr('SH_RR', Format_SR, '0010101 ----- ----- 011 ----- 0101011', L='cv.sh', fast_handler=True), + Instr('SW_RR', Format_SR, '0010110 ----- ----- 011 ----- 0101011', L='cv.sw', fast_handler=True), + ] + + # --- Custom-1 Plane A (0x2B, funct3=011): General ALU Operations --- + instrs += [ + Instr('p.ror', Format_R, '0100000 ----- ----- 011 ----- 0101011', L='cv.ror'), + Instr('p.ff1', Format_R1, '0100001 00000 ----- 011 ----- 0101011', L='cv.ff1'), + Instr('p.fl1', Format_R1, '0100010 00000 ----- 011 ----- 0101011', L='cv.fl1'), + Instr('p.clb', Format_R1, '0100011 00000 ----- 011 ----- 0101011', L='cv.clb'), + Instr('p.cnt', Format_R1, '0100100 00000 ----- 011 ----- 0101011', L='cv.cnt'), + Instr('p.abs', Format_R1, '0101000 00000 ----- 011 ----- 0101011', L='cv.abs'), + Instr('p.slet', Format_R, '0101001 ----- ----- 011 ----- 0101011', L='cv.slet'), + Instr('p.sletu', Format_R, '0101010 ----- ----- 011 ----- 0101011', L='cv.sletu'), + Instr('p.min', Format_R, '0101011 ----- ----- 011 ----- 0101011', L='cv.min'), + Instr('p.minu', Format_R, '0101100 ----- ----- 011 ----- 0101011', L='cv.minu'), + Instr('p.max', Format_R, '0101101 ----- ----- 011 ----- 0101011', L='cv.max'), + Instr('p.maxu', Format_R, '0101110 ----- ----- 011 ----- 0101011', L='cv.maxu'), + Instr('p.exths', Format_R1, '0110000 00000 ----- 011 ----- 0101011', L='cv.exths'), + Instr('p.exthz', Format_R1, '0110001 00000 ----- 011 ----- 0101011', L='cv.exthz'), + Instr('p.extbs', Format_R1, '0110010 00000 ----- 011 ----- 0101011', L='cv.extbs'), + Instr('p.extbz', Format_R1, '0110011 00000 ----- 011 ----- 0101011', L='cv.extbz'), + Instr('p.clipi', Format_I1U, '0111000 ----- ----- 011 ----- 0101011', L='cv.clip'), + Instr('p.clipui', Format_I1U, '0111001 ----- ----- 011 ----- 0101011', L='cv.clipu'), + Instr('cv.clipr', Format_R, '0111010 ----- ----- 011 ----- 0101011'), + Instr('cv.clipur', Format_R, '0111011 ----- ----- 011 ----- 0101011'), + ] + + # --- Custom-1 Plane A (0x2B, funct3=011): Register-Register Add/Sub with Normalization --- + # funct7[31:29]=100, funct7[27]=sub, funct7[26]=round, funct7[25]=unsigned + # Shift amount from register (Format_RRRR2), handlers p_addN_exec etc. + instrs += [ + Instr('p.addN', Format_RRRR2, '1000000 ----- ----- 011 ----- 0101011', L='cv.addN'), + Instr('p.adduN', Format_RRRR2, '1000001 ----- ----- 011 ----- 0101011', L='cv.adduN'), + Instr('p.addRN', Format_RRRR2, '1000010 ----- ----- 011 ----- 0101011', L='cv.addRN'), + Instr('p.adduRN', Format_RRRR2, '1000011 ----- ----- 011 ----- 0101011', L='cv.adduRN'), + Instr('p.subN', Format_RRRR2, '1000100 ----- ----- 011 ----- 0101011', L='cv.subN'), + Instr('p.subuN', Format_RRRR2, '1000101 ----- ----- 011 ----- 0101011', L='cv.subuN'), + Instr('p.subRN', Format_RRRR2, '1000110 ----- ----- 011 ----- 0101011', L='cv.subRN'), + Instr('p.subuRN', Format_RRRR2, '1000111 ----- ----- 011 ----- 0101011', L='cv.subuRN'), + ] + + # --- Custom-1 Plane A (0x2B, funct3=011): Register Bit-Manipulation --- + # funct7[31:27]=00110 (7'b0011xxx), instr[27:25] selects variant + instrs += [ + Instr('p.extract', Format_R, '0011000 ----- ----- 011 ----- 0101011', L='cv.extractr'), + Instr('p.extractu', Format_R, '0011001 ----- ----- 011 ----- 0101011', L='cv.extractur'), + Instr('p.insert', Format_I5U2, '0011010 ----- ----- 011 ----- 0101011', L='cv.insertr'), + Instr('p.bclr', Format_R, '0011100 ----- ----- 011 ----- 0101011', L='cv.bclrr'), + Instr('p.bset', Format_R, '0011101 ----- ----- 011 ----- 0101011', L='cv.bsetr'), + ] + + # --- Custom-2 (0x5B): Bit-Manipulation, Add/Sub Norm, Multiply, MAC --- + # bits[14:13]=00: Bit manipulation (immediate) + # Selector: {bits[31:30], bit[12]} + instrs += [ + Instr('p.extracti', Format_I4U, '00----- ----- ----- 000 ----- 1011011', L='cv.extract'), + Instr('p.extractui', Format_I4U, '01----- ----- ----- 000 ----- 1011011', L='cv.extractu'), + Instr('p.inserti', Format_I5U, '10----- ----- ----- 000 ----- 1011011', L='cv.insert'), + Instr('p.bclri', Format_I4U, '00----- ----- ----- 001 ----- 1011011', L='cv.bclr'), + Instr('p.bseti', Format_I4U, '01----- ----- ----- 001 ----- 1011011', L='cv.bset'), + Instr('p.bitrev', Format_BITREV, '11000-- ----- ----- 001 ----- 1011011', L='cv.bitrev'), + ] + + # --- Custom-2 (0x5B): Add/Sub Norm, Multiply, MAC --- + # Using 'p.' prefix to use handlers from pulp_v2.hpp + # v2 encoding: opcode=0x5B, bits[31:30] select variant, funct3 selects family: + # funct3=010: addN variants funct3=011: subN variants + # funct3=100: mulsN variants funct3=101: muluN variants + # funct3=110: macsN variants funct3=111: macuN variants + # bits[31:30]: 00=base, 01=hh(MAC/MUL)/unsigned(ADD/SUB), 10=round, 11=hh+round/unsigned+round + + # Add/Sub with Normalization (funct3=010 add, 011 sub) + instrs += [ + Instr('p.addNi', Format_RRRU2,'00----- ----- ----- 010 ----- 1011011', L='cv.addN'), + Instr('p.adduNi', Format_RRRU2,'01----- ----- ----- 010 ----- 1011011', L='cv.adduN'), + Instr('p.addRNi', Format_RRRU2,'10----- ----- ----- 010 ----- 1011011', L='cv.addRN'), + Instr('p.adduRNi', Format_RRRU2,'11----- ----- ----- 010 ----- 1011011', L='cv.adduRN'), + Instr('p.subNi', Format_RRRU2,'00----- ----- ----- 011 ----- 1011011', L='cv.subN'), + Instr('p.subuNi', Format_RRRU2,'01----- ----- ----- 011 ----- 1011011', L='cv.subuN'), + Instr('p.subRNi', Format_RRRU2,'10----- ----- ----- 011 ----- 1011011', L='cv.subRN'), + Instr('p.subuRNi', Format_RRRU2,'11----- ----- ----- 011 ----- 1011011', L='cv.subuRN'), + ] + + # MUL operations (funct3=100 signed, 101 unsigned) + instrs += [ + Instr('p.mulsN', Format_RRRRU,'00----- ----- ----- 100 ----- 1011011', L='cv.mulsN'), + Instr('p.mulhhsN', Format_RRRRU,'01----- ----- ----- 100 ----- 1011011', L='cv.mulhhsN'), + Instr('p.mulsNR', Format_RRRRU,'10----- ----- ----- 100 ----- 1011011', L='cv.mulsRN'), + Instr('p.mulhhsNR', Format_RRRRU,'11----- ----- ----- 100 ----- 1011011', L='cv.mulhhsRN'), + Instr('p.muluN', Format_RRRRU,'00----- ----- ----- 101 ----- 1011011', L='cv.muluN'), + Instr('p.mulhhuN', Format_RRRRU,'01----- ----- ----- 101 ----- 1011011', L='cv.mulhhuN'), + Instr('p.muluNR', Format_RRRRU,'10----- ----- ----- 101 ----- 1011011', L='cv.muluRN'), + Instr('p.mulhhuNR', Format_RRRRU,'11----- ----- ----- 101 ----- 1011011', L='cv.mulhhuRN'), + ] + + # MAC operations (funct3=110 signed, 111 unsigned) + instrs += [ + Instr('p.macsN', Format_RRRRU,'00----- ----- ----- 110 ----- 1011011', L='cv.macsN'), + Instr('p.machhsN', Format_RRRRU,'01----- ----- ----- 110 ----- 1011011', L='cv.machhsN'), + Instr('p.macsNR', Format_RRRRU,'10----- ----- ----- 110 ----- 1011011', L='cv.macsRN'), + Instr('p.machhsNR', Format_RRRRU,'11----- ----- ----- 110 ----- 1011011', L='cv.machhsRN'), + Instr('p.macuN', Format_RRRRU,'00----- ----- ----- 111 ----- 1011011', L='cv.macuN'), + Instr('p.machhuN', Format_RRRRU,'01----- ----- ----- 111 ----- 1011011', L='cv.machhuN'), + Instr('p.macuNR', Format_RRRRU,'10----- ----- ----- 111 ----- 1011011', L='cv.macuRN'), + Instr('p.machhuNR', Format_RRRRU,'11----- ----- ----- 111 ----- 1011011', L='cv.machhuRN'), + ] + + # HW loops (custom-1 0x2B, funct3=100): instr[11:8] selects the instruction, + # instr[7] selects the loop number (0 or 1) + instrs += [ + Instr('lp.starti',Format_HL0,'------- ----- 00000 100 0000- 0101011', L='cv.starti'), + Instr('lp.start', Format_HL0,'0000000 00000 ----- 100 0001- 0101011', L='cv.start'), + Instr('lp.endi', Format_HL0,'------- ----- 00000 100 0010- 0101011', L='cv.endi'), + Instr('lp.end', Format_HL0,'0000000 00000 ----- 100 0011- 0101011', L='cv.end'), + Instr('lp.counti',Format_HL0,'------- ----- 00000 100 0100- 0101011', L='cv.counti'), + Instr('lp.count', Format_HL0,'0000000 00000 ----- 100 0101- 0101011', L='cv.count'), + Instr('lp.setupi',Format_HL1,'------- ----- ----- 100 0110- 0101011', L='cv.setupi'), + Instr('lp.setup', Format_HL0,'------- ----- ----- 100 0111- 0101011', L='cv.setup'), + ] + + # mac/msu (custom-1 0x2B, funct3=011), reusing the PULP handlers: + # cv.mac funct7=1001000, cv.msu funct7=1001001 + instrs += [ + Instr('p.mac', Format_RRRR, '1001000 ----- ----- 011 ----- 0101011', L='cv.mac'), + Instr('p.msu', Format_RRRR, '1001001 ----- ----- 011 ----- 0101011', L='cv.msu'), + ] + + + # SIMD decode table: PulpV2 handlers rebased to opcode 0x7B (custom-3) + instrs += [ + Instr('pv.abs.b', Format_R1, '0111000 00000 ----- 001 ----- 1111011', L='cv.abs.b'), + Instr('pv.abs.h', Format_R1, '0111000 00000 ----- 000 ----- 1111011', L='cv.abs.h'), + Instr('pv.add.sci.b', Format_RRS, '000000- ----- ----- 111 ----- 1111011', L='cv.add.sci.b'), + Instr('pv.add.sci.h', Format_RRS, '000000- ----- ----- 110 ----- 1111011', L='cv.add.sci.h'), + Instr('pv.and.b', Format_R, '0110100 ----- ----- 001 ----- 1111011', L='cv.and.b'), + Instr('pv.and.h', Format_R, '0110100 ----- ----- 000 ----- 1111011', L='cv.and.h'), + Instr('pv.and.sc.b', Format_R, '0110100 ----- ----- 101 ----- 1111011', L='cv.and.sc.b'), + Instr('pv.and.sc.h', Format_R, '0110100 ----- ----- 100 ----- 1111011', L='cv.and.sc.h'), + Instr('pv.and.sci.b', Format_RRS, '011010- ----- ----- 111 ----- 1111011', L='cv.and.sci.b'), + Instr('pv.and.sci.h', Format_RRS, '011010- ----- ----- 110 ----- 1111011', L='cv.and.sci.h'), + Instr('pv.avg.b', Format_R, '0001000 ----- ----- 001 ----- 1111011', L='cv.avg.b'), + Instr('pv.avg.h', Format_R, '0001000 ----- ----- 000 ----- 1111011', L='cv.avg.h'), + Instr('pv.avg.sc.b', Format_R, '0001000 ----- ----- 101 ----- 1111011', L='cv.avg.sc.b'), + Instr('pv.avg.sc.h', Format_R, '0001000 ----- ----- 100 ----- 1111011', L='cv.avg.sc.h'), + Instr('pv.avg.sci.b', Format_RRS, '000100- ----- ----- 111 ----- 1111011', L='cv.avg.sci.b'), + Instr('pv.avg.sci.h', Format_RRS, '000100- ----- ----- 110 ----- 1111011', L='cv.avg.sci.h'), + Instr('pv.avgu.b', Format_R, '0001100 ----- ----- 001 ----- 1111011', L='cv.avgu.b'), + Instr('pv.avgu.h', Format_R, '0001100 ----- ----- 000 ----- 1111011', L='cv.avgu.h'), + Instr('pv.avgu.sc.b', Format_R, '0001100 ----- ----- 101 ----- 1111011', L='cv.avgu.sc.b'), + Instr('pv.avgu.sc.h', Format_R, '0001100 ----- ----- 100 ----- 1111011', L='cv.avgu.sc.h'), + Instr('pv.avgu.sci.b', Format_RRU, '000110- ----- ----- 111 ----- 1111011', L='cv.avgu.sci.b'), + Instr('pv.avgu.sci.h', Format_RRU, '000110- ----- ----- 110 ----- 1111011', L='cv.avgu.sci.h'), + Instr('pv.cmpeq.b', Format_R, '0000010 ----- ----- 001 ----- 1111011', L='cv.cmpeq.b'), + Instr('pv.cmpeq.h', Format_R, '0000010 ----- ----- 000 ----- 1111011', L='cv.cmpeq.h'), + Instr('pv.cmpeq.sc.b', Format_R, '0000010 ----- ----- 101 ----- 1111011', L='cv.cmpeq.sc.b'), + Instr('pv.cmpeq.sc.h', Format_R, '0000010 ----- ----- 100 ----- 1111011', L='cv.cmpeq.sc.h'), + Instr('pv.cmpeq.sci.b', Format_RRS, '000001- ----- ----- 111 ----- 1111011', L='cv.cmpeq.sci.b'), + Instr('pv.cmpeq.sci.h', Format_RRS, '000001- ----- ----- 110 ----- 1111011', L='cv.cmpeq.sci.h'), + Instr('pv.cmpge.b', Format_R, '0001110 ----- ----- 001 ----- 1111011', L='cv.cmpge.b'), + Instr('pv.cmpge.h', Format_R, '0001110 ----- ----- 000 ----- 1111011', L='cv.cmpge.h'), + Instr('pv.cmpge.sc.b', Format_R, '0001110 ----- ----- 101 ----- 1111011', L='cv.cmpge.sc.b'), + Instr('pv.cmpge.sc.h', Format_R, '0001110 ----- ----- 100 ----- 1111011', L='cv.cmpge.sc.h'), + Instr('pv.cmpge.sci.b', Format_RRS, '000111- ----- ----- 111 ----- 1111011', L='cv.cmpge.sci.b'), + Instr('pv.cmpge.sci.h', Format_RRS, '000111- ----- ----- 110 ----- 1111011', L='cv.cmpge.sci.h'), + Instr('pv.cmpgeu.b', Format_R, '0011110 ----- ----- 001 ----- 1111011', L='cv.cmpgeu.b'), + Instr('pv.cmpgeu.h', Format_R, '0011110 ----- ----- 000 ----- 1111011', L='cv.cmpgeu.h'), + Instr('pv.cmpgeu.sc.b', Format_R, '0011110 ----- ----- 101 ----- 1111011', L='cv.cmpgeu.sc.b'), + Instr('pv.cmpgeu.sc.h', Format_R, '0011110 ----- ----- 100 ----- 1111011', L='cv.cmpgeu.sc.h'), + Instr('pv.cmpgeu.sci.b', Format_RRU, '001111- ----- ----- 111 ----- 1111011', L='cv.cmpgeu.sci.b'), + Instr('pv.cmpgeu.sci.h', Format_RRU, '001111- ----- ----- 110 ----- 1111011', L='cv.cmpgeu.sci.h'), + Instr('pv.cmpgt.b', Format_R, '0001010 ----- ----- 001 ----- 1111011', L='cv.cmpgt.b'), + Instr('pv.cmpgt.h', Format_R, '0001010 ----- ----- 000 ----- 1111011', L='cv.cmpgt.h'), + Instr('pv.cmpgt.sc.b', Format_R, '0001010 ----- ----- 101 ----- 1111011', L='cv.cmpgt.sc.b'), + Instr('pv.cmpgt.sc.h', Format_R, '0001010 ----- ----- 100 ----- 1111011', L='cv.cmpgt.sc.h'), + Instr('pv.cmpgt.sci.b', Format_RRS, '000101- ----- ----- 111 ----- 1111011', L='cv.cmpgt.sci.b'), + Instr('pv.cmpgt.sci.h', Format_RRS, '000101- ----- ----- 110 ----- 1111011', L='cv.cmpgt.sci.h'), + Instr('pv.cmpgtu.b', Format_R, '0011010 ----- ----- 001 ----- 1111011', L='cv.cmpgtu.b'), + Instr('pv.cmpgtu.h', Format_R, '0011010 ----- ----- 000 ----- 1111011', L='cv.cmpgtu.h'), + Instr('pv.cmpgtu.sc.b', Format_R, '0011010 ----- ----- 101 ----- 1111011', L='cv.cmpgtu.sc.b'), + Instr('pv.cmpgtu.sc.h', Format_R, '0011010 ----- ----- 100 ----- 1111011', L='cv.cmpgtu.sc.h'), + Instr('pv.cmpgtu.sci.b', Format_RRU, '001101- ----- ----- 111 ----- 1111011', L='cv.cmpgtu.sci.b'), + Instr('pv.cmpgtu.sci.h', Format_RRU, '001101- ----- ----- 110 ----- 1111011', L='cv.cmpgtu.sci.h'), + Instr('pv.cmple.b', Format_R, '0010110 ----- ----- 001 ----- 1111011', L='cv.cmple.b'), + Instr('pv.cmple.h', Format_R, '0010110 ----- ----- 000 ----- 1111011', L='cv.cmple.h'), + Instr('pv.cmple.sc.b', Format_R, '0010110 ----- ----- 101 ----- 1111011', L='cv.cmple.sc.b'), + Instr('pv.cmple.sc.h', Format_R, '0010110 ----- ----- 100 ----- 1111011', L='cv.cmple.sc.h'), + Instr('pv.cmple.sci.b', Format_RRS, '001011- ----- ----- 111 ----- 1111011', L='cv.cmple.sci.b'), + Instr('pv.cmple.sci.h', Format_RRS, '001011- ----- ----- 110 ----- 1111011', L='cv.cmple.sci.h'), + Instr('pv.cmpleu.b', Format_R, '0100110 ----- ----- 001 ----- 1111011', L='cv.cmpleu.b'), + Instr('pv.cmpleu.h', Format_R, '0100110 ----- ----- 000 ----- 1111011', L='cv.cmpleu.h'), + Instr('pv.cmpleu.sc.b', Format_R, '0100110 ----- ----- 101 ----- 1111011', L='cv.cmpleu.sc.b'), + Instr('pv.cmpleu.sc.h', Format_R, '0100110 ----- ----- 100 ----- 1111011', L='cv.cmpleu.sc.h'), + Instr('pv.cmpleu.sci.b', Format_RRU, '010011- ----- ----- 111 ----- 1111011', L='cv.cmpleu.sci.b'), + Instr('pv.cmpleu.sci.h', Format_RRU, '010011- ----- ----- 110 ----- 1111011', L='cv.cmpleu.sci.h'), + Instr('pv.cmplt.b', Format_R, '0010010 ----- ----- 001 ----- 1111011', L='cv.cmplt.b'), + Instr('pv.cmplt.h', Format_R, '0010010 ----- ----- 000 ----- 1111011', L='cv.cmplt.h'), + Instr('pv.cmplt.sc.b', Format_R, '0010010 ----- ----- 101 ----- 1111011', L='cv.cmplt.sc.b'), + Instr('pv.cmplt.sc.h', Format_R, '0010010 ----- ----- 100 ----- 1111011', L='cv.cmplt.sc.h'), + Instr('pv.cmplt.sci.b', Format_RRS, '001001- ----- ----- 111 ----- 1111011', L='cv.cmplt.sci.b'), + Instr('pv.cmplt.sci.h', Format_RRS, '001001- ----- ----- 110 ----- 1111011', L='cv.cmplt.sci.h'), + Instr('pv.cmpltu.b', Format_R, '0100010 ----- ----- 001 ----- 1111011', L='cv.cmpltu.b'), + Instr('pv.cmpltu.h', Format_R, '0100010 ----- ----- 000 ----- 1111011', L='cv.cmpltu.h'), + Instr('pv.cmpltu.sc.b', Format_R, '0100010 ----- ----- 101 ----- 1111011', L='cv.cmpltu.sc.b'), + Instr('pv.cmpltu.sc.h', Format_R, '0100010 ----- ----- 100 ----- 1111011', L='cv.cmpltu.sc.h'), + Instr('pv.cmpltu.sci.b', Format_RRU, '010001- ----- ----- 111 ----- 1111011', L='cv.cmpltu.sci.b'), + Instr('pv.cmpltu.sci.h', Format_RRU, '010001- ----- ----- 110 ----- 1111011', L='cv.cmpltu.sci.h'), + Instr('pv.cmpne.b', Format_R, '0000110 ----- ----- 001 ----- 1111011', L='cv.cmpne.b'), + Instr('pv.cmpne.h', Format_R, '0000110 ----- ----- 000 ----- 1111011', L='cv.cmpne.h'), + Instr('pv.cmpne.sc.b', Format_R, '0000110 ----- ----- 101 ----- 1111011', L='cv.cmpne.sc.b'), + Instr('pv.cmpne.sc.h', Format_R, '0000110 ----- ----- 100 ----- 1111011', L='cv.cmpne.sc.h'), + Instr('pv.cmpne.sci.b', Format_RRS, '000011- ----- ----- 111 ----- 1111011', L='cv.cmpne.sci.b'), + Instr('pv.cmpne.sci.h', Format_RRS, '000011- ----- ----- 110 ----- 1111011', L='cv.cmpne.sci.h'), + Instr('pv.dotup.b', Format_R, '1000000 ----- ----- 001 ----- 1111011', L='cv.dotup.b'), + Instr('pv.dotup.h', Format_R, '1000000 ----- ----- 000 ----- 1111011', L='cv.dotup.h'), + Instr('pv.dotup.b.sc', Format_R, '1000000 ----- ----- 101 ----- 1111011', L='cv.dotup.sc.b'), + Instr('pv.dotup.h.sc', Format_R, '1000000 ----- ----- 100 ----- 1111011', L='cv.dotup.sc.h'), + Instr('pv.dotup.b.sci', Format_RRU, '100000- ----- ----- 111 ----- 1111011', L='cv.dotup.sci.b'), + Instr('pv.dotup.h.sci', Format_RRU, '100000- ----- ----- 110 ----- 1111011', L='cv.dotup.sci.h'), + Instr('pv.dotusp.b', Format_R, '1000100 ----- ----- 001 ----- 1111011', L='cv.dotusp.b'), + Instr('pv.dotusp.h', Format_R, '1000100 ----- ----- 000 ----- 1111011', L='cv.dotusp.h'), + Instr('pv.dotusp.b.sc', Format_R, '1000100 ----- ----- 101 ----- 1111011', L='cv.dotusp.sc.b'), + Instr('pv.dotusp.h.sc', Format_R, '1000100 ----- ----- 100 ----- 1111011', L='cv.dotusp.sc.h'), + Instr('pv.dotusp.b.sci', Format_RRS, '100010- ----- ----- 111 ----- 1111011', L='cv.dotusp.sci.b'), + Instr('pv.dotusp.h.sci', Format_RRS, '100010- ----- ----- 110 ----- 1111011', L='cv.dotusp.sci.h'), + Instr('pv.max.b', Format_R, '0011000 ----- ----- 001 ----- 1111011', L='cv.max.b'), + Instr('pv.max.h', Format_R, '0011000 ----- ----- 000 ----- 1111011', L='cv.max.h'), + Instr('pv.max.sc.b', Format_R, '0011000 ----- ----- 101 ----- 1111011', L='cv.max.sc.b'), + Instr('pv.max.sc.h', Format_R, '0011000 ----- ----- 100 ----- 1111011', L='cv.max.sc.h'), + Instr('pv.max.sci.b', Format_RRS, '001100- ----- ----- 111 ----- 1111011', L='cv.max.sci.b'), + Instr('pv.max.sci.h', Format_RRS, '001100- ----- ----- 110 ----- 1111011', L='cv.max.sci.h'), + Instr('pv.maxu.b', Format_R, '0011100 ----- ----- 001 ----- 1111011', L='cv.maxu.b'), + Instr('pv.maxu.h', Format_R, '0011100 ----- ----- 000 ----- 1111011', L='cv.maxu.h'), + Instr('pv.maxu.sc.b', Format_R, '0011100 ----- ----- 101 ----- 1111011', L='cv.maxu.sc.b'), + Instr('pv.maxu.sc.h', Format_R, '0011100 ----- ----- 100 ----- 1111011', L='cv.maxu.sc.h'), + Instr('pv.maxu.sci.b', Format_RRU, '001110- ----- ----- 111 ----- 1111011', L='cv.maxu.sci.b'), + Instr('pv.maxu.sci.h', Format_RRU, '001110- ----- ----- 110 ----- 1111011', L='cv.maxu.sci.h'), + Instr('pv.min.b', Format_R, '0010000 ----- ----- 001 ----- 1111011', L='cv.min.b'), + Instr('pv.min.h', Format_R, '0010000 ----- ----- 000 ----- 1111011', L='cv.min.h'), + Instr('pv.min.sc.b', Format_R, '0010000 ----- ----- 101 ----- 1111011', L='cv.min.sc.b'), + Instr('pv.min.sc.h', Format_R, '0010000 ----- ----- 100 ----- 1111011', L='cv.min.sc.h'), + Instr('pv.min.sci.b', Format_RRS, '001000- ----- ----- 111 ----- 1111011', L='cv.min.sci.b'), + Instr('pv.min.sci.h', Format_RRS, '001000- ----- ----- 110 ----- 1111011', L='cv.min.sci.h'), + Instr('pv.minu.b', Format_R, '0010100 ----- ----- 001 ----- 1111011', L='cv.minu.b'), + Instr('pv.minu.h', Format_R, '0010100 ----- ----- 000 ----- 1111011', L='cv.minu.h'), + Instr('pv.minu.sc.b', Format_R, '0010100 ----- ----- 101 ----- 1111011', L='cv.minu.sc.b'), + Instr('pv.minu.sc.h', Format_R, '0010100 ----- ----- 100 ----- 1111011', L='cv.minu.sc.h'), + Instr('pv.minu.sci.b', Format_RRU, '001010- ----- ----- 111 ----- 1111011', L='cv.minu.sci.b'), + Instr('pv.minu.sci.h', Format_RRU, '001010- ----- ----- 110 ----- 1111011', L='cv.minu.sci.h'), + Instr('pv.or.b', Format_R, '0101100 ----- ----- 001 ----- 1111011', L='cv.or.b'), + Instr('pv.or.h', Format_R, '0101100 ----- ----- 000 ----- 1111011', L='cv.or.h'), + Instr('pv.or.sc.b', Format_R, '0101100 ----- ----- 101 ----- 1111011', L='cv.or.sc.b'), + Instr('pv.or.sc.h', Format_R, '0101100 ----- ----- 100 ----- 1111011', L='cv.or.sc.h'), + Instr('pv.or.sci.b', Format_RRS, '010110- ----- ----- 111 ----- 1111011', L='cv.or.sci.b'), + Instr('pv.or.sci.h', Format_RRS, '010110- ----- ----- 110 ----- 1111011', L='cv.or.sci.h'), + Instr('pv.shuffle.b', Format_R, '1100000 ----- ----- 001 ----- 1111011', L='cv.shuffle.b'), + Instr('pv.shuffle.h', Format_R, '1100000 ----- ----- 000 ----- 1111011', L='cv.shuffle.h'), + Instr('pv.shuffle.h.sci', Format_RRU, '110000- 0000- ----- 110 ----- 1111011', L='cv.shuffle.sci.h'), + Instr('pv.shufflei0.b.sci', Format_RRU2, '110000- ----- ----- 111 ----- 1111011', L='cv.shufflei0.sci.b'), + Instr('pv.sll.b', Format_R, '0101000 ----- ----- 001 ----- 1111011', L='cv.sll.b'), + Instr('pv.sll.h', Format_R, '0101000 ----- ----- 000 ----- 1111011', L='cv.sll.h'), + Instr('pv.sll.sc.b', Format_R, '0101000 ----- ----- 101 ----- 1111011', L='cv.sll.sc.b'), + Instr('pv.sll.sc.h', Format_R, '0101000 ----- ----- 100 ----- 1111011', L='cv.sll.sc.h'), + Instr('pv.sll.sci.b', Format_RRU, '010100- 000-- ----- 111 ----- 1111011', L='cv.sll.sci.b'), + Instr('pv.sll.sci.h', Format_RRU, '010100- 00--- ----- 110 ----- 1111011', L='cv.sll.sci.h'), + Instr('pv.sra.b', Format_R, '0100100 ----- ----- 001 ----- 1111011', L='cv.sra.b'), + Instr('pv.sra.h', Format_R, '0100100 ----- ----- 000 ----- 1111011', L='cv.sra.h'), + Instr('pv.sra.sc.b', Format_R, '0100100 ----- ----- 101 ----- 1111011', L='cv.sra.sc.b'), + Instr('pv.sra.sc.h', Format_R, '0100100 ----- ----- 100 ----- 1111011', L='cv.sra.sc.h'), + Instr('pv.sra.sci.b', Format_RRS, '010010- 000-- ----- 111 ----- 1111011', L='cv.sra.sci.b'), + Instr('pv.sra.sci.h', Format_RRS, '010010- 00--- ----- 110 ----- 1111011', L='cv.sra.sci.h'), + Instr('pv.srl.b', Format_R, '0100000 ----- ----- 001 ----- 1111011', L='cv.srl.b'), + Instr('pv.srl.h', Format_R, '0100000 ----- ----- 000 ----- 1111011', L='cv.srl.h'), + Instr('pv.srl.sc.b', Format_R, '0100000 ----- ----- 101 ----- 1111011', L='cv.srl.sc.b'), + Instr('pv.srl.sc.h', Format_R, '0100000 ----- ----- 100 ----- 1111011', L='cv.srl.sc.h'), + Instr('pv.srl.sci.b', Format_RRU, '010000- 000-- ----- 111 ----- 1111011', L='cv.srl.sci.b'), + Instr('pv.srl.sci.h', Format_RRU, '010000- 00--- ----- 110 ----- 1111011', L='cv.srl.sci.h'), + Instr('pv.sub.b', Format_R, '0000100 ----- ----- 001 ----- 1111011', L='cv.sub.b'), + Instr('pv.sub.h', Format_R, '0000100 ----- ----- 000 ----- 1111011', L='cv.sub.h'), + Instr('pv.sub.sc.b', Format_R, '0000100 ----- ----- 101 ----- 1111011', L='cv.sub.sc.b'), + Instr('pv.sub.sc.h', Format_R, '0000100 ----- ----- 100 ----- 1111011', L='cv.sub.sc.h'), + Instr('pv.sub.sci.b', Format_RRS, '000010- ----- ----- 111 ----- 1111011', L='cv.sub.sci.b'), + Instr('pv.sub.sci.h', Format_RRS, '000010- ----- ----- 110 ----- 1111011', L='cv.sub.sci.h'), + Instr('pv.xor.b', Format_R, '0110000 ----- ----- 001 ----- 1111011', L='cv.xor.b'), + Instr('pv.xor.h', Format_R, '0110000 ----- ----- 000 ----- 1111011', L='cv.xor.h'), + Instr('pv.xor.sc.b', Format_R, '0110000 ----- ----- 101 ----- 1111011', L='cv.xor.sc.b'), + Instr('pv.xor.sc.h', Format_R, '0110000 ----- ----- 100 ----- 1111011', L='cv.xor.sc.h'), + Instr('pv.xor.sci.b', Format_RRS, '011000- ----- ----- 111 ----- 1111011', L='cv.xor.sci.b'), + Instr('pv.xor.sci.h', Format_RRS, '011000- ----- ----- 110 ----- 1111011', L='cv.xor.sci.h'), + Instr('pv.add.h', Format_R, '0000000 ----- ----- 000 ----- 1111011', L='cv.add.h'), + Instr('pv.add.b', Format_R, '0000000 ----- ----- 001 ----- 1111011', L='cv.add.b'), + Instr('pv.add.sc.h', Format_R, '0000000 ----- ----- 100 ----- 1111011', L='cv.add.sc.h'), + Instr('pv.add.sc.b', Format_R, '0000000 ----- ----- 101 ----- 1111011', L='cv.add.sc.b'), + # extract / insert / shuffle2 / shufflei / pack + dotsp / sdot + Instr('pv.extract.h', Format_RRU, '101110- 00000 ----- 000 ----- 1111011', L='cv.extract.h'), + Instr('pv.extract.b', Format_RRU, '101110- 0000- ----- 001 ----- 1111011', L='cv.extract.b'), + Instr('pv.extractu.h', Format_RRU, '101110- 00000 ----- 010 ----- 1111011', L='cv.extractu.h'), + Instr('pv.extractu.b', Format_RRU, '101110- 0000- ----- 011 ----- 1111011', L='cv.extractu.b'), + Instr('pv.insert.h', Format_RRRU, '101110- 00000 ----- 100 ----- 1111011', L='cv.insert.h'), + Instr('pv.insert.b', Format_RRRU, '101110- 0000- ----- 101 ----- 1111011', L='cv.insert.b'), + Instr('cv.pack.h', Format_R, '1111001 ----- ----- 000 ----- 1111011', L='cv.pack.h'), + Instr('cv.packhi.b', Format_RRRR, '1111101 ----- ----- 001 ----- 1111011', L='cv.packhi.b'), + Instr('cv.packlo.b', Format_RRRR, '1111100 ----- ----- 001 ----- 1111011', L='cv.packlo.b'), + Instr('pv.shuffle2.h', Format_RRRR, '1110000 ----- ----- 000 ----- 1111011', L='cv.shuffle2.h'), + Instr('pv.shuffle2.b', Format_RRRR, '1110000 ----- ----- 001 ----- 1111011', L='cv.shuffle2.b'), + Instr('pv.shufflei1.b.sci', Format_RRU2, '110010- ----- ----- 111 ----- 1111011', L='cv.shufflei1.sci.b'), + Instr('pv.shufflei2.b.sci', Format_RRU2, '110100- ----- ----- 111 ----- 1111011', L='cv.shufflei2.sci.b'), + Instr('pv.shufflei3.b.sci', Format_RRU2, '110110- ----- ----- 111 ----- 1111011', L='cv.shufflei3.sci.b'), + Instr('pv.dotsp.b', Format_R, '1001000 ----- ----- 001 ----- 1111011', L='cv.dotsp.b'), + Instr('pv.dotsp.h', Format_R, '1001000 ----- ----- 000 ----- 1111011', L='cv.dotsp.h'), + Instr('pv.dotsp.b.sc', Format_R, '1001000 ----- ----- 101 ----- 1111011', L='cv.dotsp.sc.b'), + Instr('pv.dotsp.h.sc', Format_R, '1001000 ----- ----- 100 ----- 1111011', L='cv.dotsp.sc.h'), + Instr('pv.dotsp.b.sci', Format_RRS, '100100- ----- ----- 111 ----- 1111011', L='cv.dotsp.sci.b'), + Instr('pv.dotsp.h.sci', Format_RRS, '100100- ----- ----- 110 ----- 1111011', L='cv.dotsp.sci.h'), + Instr('pv.sdotsp.b', Format_RRRR, '1010100 ----- ----- 001 ----- 1111011', L='cv.sdotsp.b'), + Instr('pv.sdotsp.h', Format_RRRR, '1010100 ----- ----- 000 ----- 1111011', L='cv.sdotsp.h'), + Instr('pv.sdotsp.b.sc', Format_RRRR, '1010100 ----- ----- 101 ----- 1111011', L='cv.sdotsp.sc.b'), + Instr('pv.sdotsp.h.sc', Format_RRRR, '1010100 ----- ----- 100 ----- 1111011', L='cv.sdotsp.sc.h'), + Instr('pv.sdotsp.b.sci', Format_RRRS, '101010- ----- ----- 111 ----- 1111011', L='cv.sdotsp.sci.b'), + Instr('pv.sdotsp.h.sci', Format_RRRS, '101010- ----- ----- 110 ----- 1111011', L='cv.sdotsp.sci.h'), + Instr('pv.sdotup.b', Format_RRRR, '1001100 ----- ----- 001 ----- 1111011', L='cv.sdotup.b'), + Instr('pv.sdotup.h', Format_RRRR, '1001100 ----- ----- 000 ----- 1111011', L='cv.sdotup.h'), + Instr('pv.sdotup.b.sc', Format_RRRR, '1001100 ----- ----- 101 ----- 1111011', L='cv.sdotup.sc.b'), + Instr('pv.sdotup.h.sc', Format_RRRR, '1001100 ----- ----- 100 ----- 1111011', L='cv.sdotup.sc.h'), + Instr('pv.sdotup.b.sci', Format_RRRU, '100110- ----- ----- 111 ----- 1111011', L='cv.sdotup.sci.b'), + Instr('pv.sdotup.h.sci', Format_RRRU, '100110- ----- ----- 110 ----- 1111011', L='cv.sdotup.sci.h'), + Instr('pv.sdotusp.b', Format_RRRR, '1010000 ----- ----- 001 ----- 1111011', L='cv.sdotusp.b'), + Instr('pv.sdotusp.h', Format_RRRR, '1010000 ----- ----- 000 ----- 1111011', L='cv.sdotusp.h'), + Instr('pv.sdotusp.b.sc', Format_RRRR, '1010000 ----- ----- 101 ----- 1111011', L='cv.sdotusp.sc.b'), + Instr('pv.sdotusp.h.sc', Format_RRRR, '1010000 ----- ----- 100 ----- 1111011', L='cv.sdotusp.sc.h'), + Instr('pv.sdotusp.b.sci', Format_RRRS, '101000- ----- ----- 111 ----- 1111011', L='cv.sdotusp.sci.b'), + Instr('pv.sdotusp.h.sci', Format_RRRS, '101000- ----- ----- 110 ----- 1111011', L='cv.sdotusp.sci.h'), + # complex multiply / div-rounding / pack + Instr('cv.add.div2', Format_R, '0110110 ----- ----- 010 ----- 1111011', L='cv.add.div2'), + Instr('cv.add.div4', Format_R, '0110110 ----- ----- 100 ----- 1111011', L='cv.add.div4'), + Instr('cv.add.div8', Format_R, '0110110 ----- ----- 110 ----- 1111011', L='cv.add.div8'), + Instr('cv.sub.div2', Format_R, '0111010 ----- ----- 010 ----- 1111011', L='cv.sub.div2'), + Instr('cv.sub.div4', Format_R, '0111010 ----- ----- 100 ----- 1111011', L='cv.sub.div4'), + Instr('cv.sub.div8', Format_R, '0111010 ----- ----- 110 ----- 1111011', L='cv.sub.div8'), + Instr('cv.subrotmj', Format_R, '0110010 ----- ----- 000 ----- 1111011', L='cv.subrotmj'), + Instr('cv.subrotmj.div2', Format_R, '0110010 ----- ----- 010 ----- 1111011', L='cv.subrotmj.div2'), + Instr('cv.subrotmj.div4', Format_R, '0110010 ----- ----- 100 ----- 1111011', L='cv.subrotmj.div4'), + Instr('cv.subrotmj.div8', Format_R, '0110010 ----- ----- 110 ----- 1111011', L='cv.subrotmj.div8'), + Instr('cv.cplxconj', Format_R1, '0101110 00000 ----- 000 ----- 1111011', L='cv.cplxconj'), + Instr('cv.cplxmul.r', Format_RRRR, '0101010 ----- ----- 000 ----- 1111011', L='cv.cplxmul.r'), + Instr('cv.cplxmul.r.div2', Format_RRRR, '0101010 ----- ----- 010 ----- 1111011', L='cv.cplxmul.r.div2'), + Instr('cv.cplxmul.r.div4', Format_RRRR, '0101010 ----- ----- 100 ----- 1111011', L='cv.cplxmul.r.div4'), + Instr('cv.cplxmul.r.div8', Format_RRRR, '0101010 ----- ----- 110 ----- 1111011', L='cv.cplxmul.r.div8'), + Instr('cv.cplxmul.i', Format_RRRR, '0101011 ----- ----- 000 ----- 1111011', L='cv.cplxmul.i'), + Instr('cv.cplxmul.i.div2', Format_RRRR, '0101011 ----- ----- 010 ----- 1111011', L='cv.cplxmul.i.div2'), + Instr('cv.cplxmul.i.div4', Format_RRRR, '0101011 ----- ----- 100 ----- 1111011', L='cv.cplxmul.i.div4'), + Instr('cv.cplxmul.i.div8', Format_RRRR, '0101011 ----- ----- 110 ----- 1111011', L='cv.cplxmul.i.div8'), + Instr('cv.pack', Format_R, '1111000 ----- ----- 000 ----- 1111011', L='cv.pack'), + ] + # The handler header is declared here (like PulpV2 does) so iss_v2 + # builds, which only pull in what each subset/module declares, get + # the exec functions. v1 builds already include it via the core's + # class.hpp; the double include is guarded. + super().__init__(name='pulpv2', instrs=instrs, includes=[ + '', + ]) diff --git a/models/cpu/iss/isa_gen/isa_gen.py b/models/cpu/iss/isa_gen/isa_gen.py index a58367fa1..bc25509ce 100644 --- a/models/cpu/iss/isa_gen/isa_gen.py +++ b/models/cpu/iss/isa_gen/isa_gen.py @@ -972,6 +972,16 @@ def gen(self, isa, isaFile, opcode, others=False): for tag_name in isa.isa_tags_insns.keys(): insn_tags_value.append('1' if tag_name in self.tags else '0') + # Inactive instructions never decode nor execute, so emit NULL instead + # of referencing handlers whose subset header may not be included (e.g. + # the compressed FP rows when the F extension is not part of the ISA). + if self.active: + handler = self.exec_func + fast_handler = self.exec_func_fast + decode = "NULL" if self.decode is None else self.decode + else: + handler = fast_handler = decode = "NULL" + dump(isaFile, f'static iss_decoder_item_t {name} = {{\n') dump(isaFile, f' .is_insn=true,\n') dump(isaFile, f' .is_active={ 1 if self.active else 0},\n') @@ -979,10 +989,10 @@ def gen(self, isa, isaFile, opcode, others=False): dump(isaFile, f' .opcode=0b{opcode},\n') dump(isaFile, f' .u={{\n') dump(isaFile, f' .insn={{\n') - dump(isaFile, f' .handler={self.exec_func},\n') - dump(isaFile, f' .fast_handler={self.exec_func_fast},\n') + dump(isaFile, f' .handler={handler},\n') + dump(isaFile, f' .fast_handler={fast_handler},\n') dump(isaFile, f' .stub_handler=NULL,\n') - dump(isaFile, f' .decode={"NULL" if self.decode is None else self.decode},\n') + dump(isaFile, f' .decode={decode},\n') dump(isaFile, f' .label=(char *)"{self.get_label()}",\n') dump(isaFile, f' .size={int(self.len/8)},\n') dump(isaFile, f' .nb_args={len(self.args_format)},\n') diff --git a/models/cpu/iss/src/core.cpp b/models/cpu/iss/src/core.cpp index 65d1cb25a..00b2f694d 100644 --- a/models/cpu/iss/src/core.cpp +++ b/models/cpu/iss/src/core.cpp @@ -56,6 +56,11 @@ void Core::build() this->iss.csr.mstatus.reset_val |= 2ULL << 34; #endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Default: no-op. Subclasses can override to apply a core-specific mask. + this->mstatus_write_mask_fixup(); + +#endif #if ISS_REG_WIDTH == 64 this->sstatus_write_mask = this->mstatus_write_mask & 0x80000003000DE762; #else @@ -96,20 +101,39 @@ void Core::reset(bool active) } +#ifdef CONFIG_GVSOC_ISS_CV32E40P +void Core::mret_mode_restore() +{ + this->mode_set(this->iss.csr.mstatus.mpp); +#ifdef CONFIG_GVSOC_ISS_USER_MODE + this->iss.csr.mstatus.mpp = PRIV_U; +#else + this->iss.csr.mstatus.mpp = PRIV_M; +#endif +} + +#endif iss_reg_t Core::mret_handle() { this->iss.exec.switch_to_full_mode(); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Privilege-mode restore on MRET (default: from mstatus.mpp). + this->mret_mode_restore(); +#else this->mode_set(this->iss.csr.mstatus.mpp); #ifdef CONFIG_GVSOC_ISS_USER_MODE this->iss.csr.mstatus.mpp = PRIV_U; #else this->iss.csr.mstatus.mpp = PRIV_M; +#endif #endif this->iss.irq.irq_enable.set(this->iss.csr.mstatus.mpie); this->iss.csr.mstatus.mie = this->iss.csr.mstatus.mpie; this->iss.csr.mstatus.mpie = 1; +#ifndef CONFIG_GVSOC_ISS_CV32E40P this->iss.csr.mcause.value = 0; +#endif return this->iss.csr.mepc.value; } @@ -129,7 +153,9 @@ iss_reg_t Core::sret_handle() this->iss.irq.irq_enable.set(this->iss.csr.mstatus.spie); this->iss.csr.mstatus.sie = this->iss.csr.mstatus.spie; this->iss.csr.mstatus.spie = 1; +#ifndef CONFIG_GVSOC_ISS_CV32E40P this->iss.csr.scause.value = 0; +#endif return this->iss.csr.sepc.value; } @@ -167,6 +193,9 @@ bool Core::mstatus_update(bool is_write, iss_reg_t &value) else { value = this->iss.csr.mstatus.value; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + this->iss.csr.mstatus_read_fixup(value); +#endif } return false; diff --git a/models/cpu/iss/src/csr.cpp b/models/cpu/iss/src/csr.cpp index cb90ce710..6d95dd233 100644 --- a/models/cpu/iss/src/csr.cpp +++ b/models/cpu/iss/src/csr.cpp @@ -71,6 +71,20 @@ Csr::Csr(Iss &iss) this->declare_csr(&this->tdata2, "tdata2", 0x7A2); this->declare_csr(&this->tdata3, "tdata3", 0x7A3); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Optional extension CSRs: without this config they remain unregistered + // and access falls through to the unsupported-CSR path (upstream behavior). + this->declare_csr(&this->tinfo, "tinfo", 0x7A4); + this->declare_csr(&this->mcontext, "mcontext", 0x7A8); + this->declare_csr(&this->scontext, "scontext", 0x7AA); + + this->declare_csr(&this->minstret, "minstret", 0xB02); +#if ISS_REG_WIDTH == 32 + this->declare_csr(&this->mcycleh, "mcycleh", 0xB80); + this->declare_csr(&this->minstreth, "minstreth", 0xB82); +#endif + +#endif /* CONFIG_GVSOC_ISS_CV32E40P */ // Machine Non-Maskable Interrupt Handling this->declare_csr(&this->mnscratch, "mnscratch", 0x740); this->declare_csr(&this->mnepc, "mnepc", 0x741); @@ -96,6 +110,10 @@ Csr::Csr(Iss &iss) // Machine information registers this->declare_csr(&this->mvendorid, "mvendorid", 0xF11); this->declare_csr(&this->marchid, "marchid", 0xF12); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + this->declare_csr(&this->mimpid, "mimpid", 0xF13); + this->declare_csr(&this->mhartid, "mhartid", 0xF14); +#endif this->declare_csr(&this->vstart, "vstart", 0x008, 0); this->declare_csr(&this->vxstat, "vxstat", 0x009); @@ -210,7 +228,11 @@ void Csr::build() this->iss.top.new_master_port("time", &this->time_itf, (vp::Block *)this); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + this->mhartid.value = (this->iss.top.get_js_config()->get_child_int("cluster_id") << 5) | this->iss.top.get_js_config()->get_child_int("core_id"); +#else this->mhartid = (this->iss.top.get_js_config()->get_child_int("cluster_id") << 5) | this->iss.top.get_js_config()->get_child_int("core_id"); +#endif this->tselect.register_callback(std::bind(&Csr::tselect_access, this, std::placeholders::_1, std::placeholders::_2)); @@ -224,7 +246,11 @@ bool Csr::tselect_access(bool is_write, iss_reg_t &value) { if (!is_write) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + value = this->tselect_default_read; +#else value = -1; +#endif } return false; } @@ -237,6 +263,13 @@ bool Csr::time_access(bool is_write, iss_reg_t &value) return false; } +#ifdef CONFIG_GVSOC_ISS_CV32E40P +bool Csr::mstatus_access(bool is_write, iss_reg_t &value) +{ + return true; +} + +#endif bool Csr::mcycle_access(bool is_write, iss_reg_t &value) { if (!is_write) @@ -351,37 +384,91 @@ static bool uip_write(Iss *iss, unsigned int value) static bool fflags_read(Iss *iss, iss_reg_t *value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } +#endif *value = iss->csr.fcsr.fflags; return false; } static bool fflags_write(Iss *iss, unsigned int value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } +#endif iss->csr.fcsr.fflags = value; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // RTL: fflags write (fflags_we_i) forces mstatus.FS=Dirty. + iss->csr.fp_state_dirty(); +#endif return false; } static bool frm_read(Iss *iss, iss_reg_t *value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } +#endif *value = iss->csr.fcsr.frm; return false; } static bool frm_write(Iss *iss, unsigned int value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } +#endif iss->csr.fcsr.frm = value; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // RTL: FP-CSR write (fcsr_update covers frm) forces mstatus.FS=Dirty. + iss->csr.fp_state_dirty(); +#endif return false; } static bool fcsr_read(Iss *iss, iss_reg_t *value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } +#endif *value = iss->csr.fcsr.raw; return false; } static bool fcsr_write(Iss *iss, unsigned int value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } +#endif iss->csr.fcsr.raw = value & 0xff; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // RTL: FP-CSR write (fcsr_update) forces mstatus.FS=Dirty. + iss->csr.fp_state_dirty(); +#endif return false; } @@ -437,13 +524,21 @@ static bool scounteren_write(Iss *iss, unsigned int value) static bool mimpid_read(Iss *iss, iss_reg_t *value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + *value = iss->csr.mimpid.value; +#else *value = 0; +#endif return false; } static bool mhartid_read(Iss *iss, iss_reg_t *value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + *value = iss->csr.mhartid.value; +#else *value = iss->csr.mhartid; +#endif return false; } @@ -696,6 +791,16 @@ static bool hwloop_read(Iss *iss, int reg, iss_reg_t *value) static bool hwloop_write(Iss *iss, int reg, unsigned int value) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // CV32E40P: lpstart/lpend hold a word address with the low 2 bits hardwired 0. + // Mask only those; lpcount is a count, not an address, so leave it untouched. + if (reg == PULPV2_HWLOOP_LPSTART(0) || reg == PULPV2_HWLOOP_LPEND(0) || + reg == PULPV2_HWLOOP_LPSTART(1) || reg == PULPV2_HWLOOP_LPEND(1)) + { + value &= ~0x3u; + } +#endif + iss->csr.hwloop_regs[reg] = value; // Since the HW loop is using decode instruction for the HW loop start to jump faster @@ -891,6 +996,16 @@ bool iss_csr_read(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t *value) } #endif +#if defined(CONFIG_GVSOC_ISS_CV32E40P) && (defined(CONFIG_GVSOC_ISS_RI5KY) || defined(CONFIG_GVSOC_ISS_HWLOOP)) + // HWLOOP CSR mapping (matches hwloop_read gate above). + { + int hwloop_idx = iss->csr.hwloop_csr_index(reg); + if (hwloop_idx >= 0) + { + return hwloop_read(iss, hwloop_idx, value); + } + } +#endif // New generic way of handling CSR access, all CSR should be accessed there if (!iss->csr.access(false, reg, *value)) { @@ -901,6 +1016,7 @@ bool iss_csr_read(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t *value) switch (reg) { +#ifndef CONFIG_GVSOC_ISS_CV32E40P // User trap setup case 0x000: status = ustatus_read(iss, value); @@ -928,6 +1044,7 @@ bool iss_csr_read(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t *value) case 0x044: status = uip_read(iss, value); break; +#endif // User floating-point CSRs case 0x001: @@ -1096,9 +1213,11 @@ bool iss_csr_read(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t *value) status = mhartid_read(iss, value); break; +#ifndef CONFIG_GVSOC_ISS_CV32E40P case 0x306: status = mcounteren_read(iss, value); break; +#endif // Machine timers and counters @@ -1260,9 +1379,22 @@ bool iss_csr_read(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t *value) if (status) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Config field decides exception vs warning per core. + if (iss->csr.raise_on_unsupported_csr_flag) + { + iss->csr.trace.msg(vp::Trace::LEVEL_DEBUG, "Unsupported CSR read (id: 0x%x) -> illegal instruction\n", reg); + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + } + else + { + iss->csr.trace.force_warning("Accessing unsupported CSR (id: 0x%x, name: %s)\n", reg, iss_csr_name(iss, reg)); + } +#else iss->csr.trace.force_warning("Accessing unsupported CSR (id: 0x%x, name: %s)\n", reg, iss_csr_name(iss, reg)); #if 0 triggerException_cause(iss, iss->currentPc, EXCEPTION_ILLEGAL_INSTR, ECAUSE_ILL_INSTR); +#endif #endif return true; } @@ -1278,6 +1410,16 @@ bool iss_csr_write(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t value) iss->csr.trace.msg("Writing CSR (reg: 0x%x, name: %s, value: 0x%x)\n", reg, iss_csr_name(iss, reg), value); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // HWLOOP CSRs are not writable via csrrw on CV32E40P: raise illegal. + if (iss->csr.hwloop_csr_index(reg) >= 0) + { + iss->csr.trace.msg("Illegal CSR write to hwloop register (id: 0x%x)\n", reg); + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return true; + } + +#endif // If there is any write to a CSR, switch to full check instruction handler // in case something special happened (like HW counting become active) iss->exec.switch_to_full_mode(); @@ -1301,6 +1443,7 @@ bool iss_csr_write(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t value) switch (reg) { +#ifndef CONFIG_GVSOC_ISS_CV32E40P // User trap setup case 0x000: return ustatus_write(iss, value); @@ -1320,6 +1463,7 @@ bool iss_csr_write(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t value) return ubadaddr_write(iss, value); case 0x044: return uip_write(iss, value); +#endif // User floating-point CSRs case 0x001: @@ -1389,9 +1533,22 @@ bool iss_csr_write(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t value) return hwloop_write(iss, reg - CSR_HWLOOP0_START, value); #endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Config field decides exception vs warning per core. + if (iss->csr.raise_on_unsupported_csr_flag) + { + iss->csr.trace.msg(vp::Trace::LEVEL_DEBUG, "Unsupported CSR write (id: 0x%x) -> illegal instruction\n", reg); + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + } + else + { + iss->csr.trace.force_warning("Accessing unsupported CSR (id: 0x%x, name: %s)\n", reg, iss_csr_name(iss, reg)); + } +#else iss->csr.trace.force_warning("Accessing unsupported CSR (id: 0x%x, name: %s)\n", reg, iss_csr_name(iss, reg)); #if 0 triggerException_cause(iss, iss->currentPc, EXCEPTION_ILLEGAL_INSTR, ECAUSE_ILL_INSTR); +#endif #endif return true; @@ -1709,6 +1866,13 @@ const char *iss_csr_name(Iss *iss, iss_reg_t reg) } #endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Core-specific CSR name lookup (default: fall through to "unknown"). + if (const char *core_name = iss->csr.custom_csr_name(reg)) + { + return core_name; + } +#endif return "unknown"; } diff --git a/models/cpu/iss/src/cv32e40p/core_cv32e40p.cpp b/models/cpu/iss/src/cv32e40p/core_cv32e40p.cpp new file mode 100644 index 000000000..360384dd5 --- /dev/null +++ b/models/cpu/iss/src/cv32e40p/core_cv32e40p.cpp @@ -0,0 +1,43 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#ifdef CONFIG_GVSOC_ISS_CV32E40P + +#include +#include + +// CV32E40P is M-mode only +void Cv32e40pCore::mret_mode_restore() +{ + this->mode_set(PRIV_M); + this->iss.csr.mstatus.mpp = PRIV_M; +} + +// only MIE(3) + MPIE(7) writable; MPP forced to M by hardware. +// With FPU: add FS(14:13). +void Cv32e40pCore::mstatus_write_mask_fixup() +{ + bool fpu_in_isa = this->iss.top.get_js_config()->get_child_bool("fpu_in_isa"); + this->mstatus_write_mask = fpu_in_isa ? (iss_reg_t)0x6088 : (iss_reg_t)0x0088; +} + +#endif /* CONFIG_GVSOC_ISS_CV32E40P */ diff --git a/models/cpu/iss/src/cv32e40p/csr_cv32e40p.cpp b/models/cpu/iss/src/cv32e40p/csr_cv32e40p.cpp new file mode 100644 index 000000000..8067d06c1 --- /dev/null +++ b/models/cpu/iss/src/cv32e40p/csr_cv32e40p.cpp @@ -0,0 +1,402 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * CV32E40P-specific CSR subclass implementation. + * + * + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#ifdef CONFIG_GVSOC_ISS_CV32E40P + +#include "cpu/iss/include/iss.hpp" + +// Helper: read config int with fallback default. +// Uses get() to distinguish "key missing" from "key=0". +static inline int cfg_int_or(js::Config *cfg, const char *key, int dflt) +{ + js::Config *child = cfg->get(key); + return child ? child->get_int() : dflt; +} + +Cv32e40pCsr::Cv32e40pCsr(Iss &iss) + : Csr(iss) +{ + // All CSR customization moved to build() — constructor runs too early + // during Iss member initialization (iss.top not yet set). +} + +void Cv32e40pCsr::build() +{ + Csr::build(); // Base class CSR setup (regs map, trace, etc.) + this->build_cv32e40p(); // CV32E40P-specific customization +} + +void Cv32e40pCsr::build_cv32e40p() +{ + /* reset() runs AFTER build() in GVSOC — reset_val must be correct */ + + // Cache FPU/ZFINX config for hot-path use + this->m_fpu_in_isa = this->iss.top.get_js_config()->get_child_bool("fpu_in_isa"); + this->m_zfinx = this->iss.top.get_js_config()->get_child_bool("zfinx"); + + // CV32E40P mstatus access callback — forces MPP=M on reads, lets writes + // proceed via Cv32e40pCsr::mstatus_access() override. + this->mstatus.register_callback(std::bind(&Cv32e40pCsr::mstatus_access, this, std::placeholders::_1, std::placeholders::_2)); + + // CV32E40P is M-mode only: remove CSRs that don't exist + uint32_t nonexistent[] = { + 0x100, 0x104, 0x105, 0x106, // sstatus, sie, stvec, scounteren + 0x140, 0x141, 0x142, 0x143, 0x144, // sscratch, sepc, scause, stval, sip + 0x302, 0x303, // medeleg, mideleg (no delegation) + 0x306, // mcounteren (no U-mode) + 0x740, 0x741, 0x742, 0x744, // NMI: mnscratch, mnepc, mncause, mnstatus + 0x008, 0x009, 0x00A, 0x00F, // Vector: vstart, vxstat, vxrm, vcsr + 0xC20, 0xC21, 0xC22, // Vector: vl, vtype, vlenb + }; + for (uint32_t addr : nonexistent) + { + this->undeclare_csr(addr); + } + + // mhpmevent CSRs — base Csr skips these for CV32E40P, declare here in build_cv32e40p() + for (int i = 0; i < 29; i++) + { + this->declare_csr(&this->mhpmevent[i], + "mhpmevent" + std::to_string(i + 3), 0x323 + i); + } + + // tinfo register — CV32E40P trigger info (read-only, value=0x4) + // base Csr now declares tinfo unconditionally (no undeclare+redeclare needed). + // reset_val and write_mask set below in this method. + + auto *cfg = this->iss.top.get_js_config(); + + // PULP custom CSRs (0xCD0-0xCD2) + // These exist when COREV_PULP=1 (all PULP configs). The RTL decoder + // (cv32e40p_decoder.sv, CSR_UHARTID/CSR_PRIVLV/CSR_ZFINX cases) raises + // illegal-instruction on any write op to them, so write_illegal (not a + // silently-ignored write mask) is the correct model. + bool pulpv2 = cfg->get_child_bool("pulpv2"); + if (pulpv2) + { + // PULP custom CSRs — reset_val MUST be passed to declare_csr() + // because Csr::reset() runs AFTER build() and overwrites .value with + // reset_val (default 0). Values set only via .value are clobbered. + + // 0xCD0 UHARTID — duplicate of mhartid, user-mode readable + iss_reg_t uhartid_val = cfg_int_or(cfg, "mhartid_value", 0); + this->declare_csr(&this->uhartid, "uhartid", 0xCD0, uhartid_val); + this->uhartid.set_write_mask(0); + this->uhartid.write_illegal = true; + + // 0xCD1 PRIVLV — current privilege level (M-mode only → always 3) + this->declare_csr(&this->privlv, "privlv", 0xCD1, 3); + this->privlv.set_write_mask(0); + this->privlv.write_illegal = true; + + // 0xCD2 ZFINX — reads 1 when ZFINX=1, 0 when FPU=0. The RTL decoder + // rejects it when FPU=1 && ZFINX=0 (illegal even on reads), so in + // that config it must stay undeclared: the unsupported-CSR path then + // raises illegal-instruction, matching the RTL. + if (!this->m_fpu_in_isa) + { + iss_reg_t zfinx_val = (this->m_zfinx) ? 1 : 0; + this->declare_csr(&this->zfinx_csr, "zfinx", 0xCD2, zfinx_val); + this->zfinx_csr.set_write_mask(0); + this->zfinx_csr.write_illegal = true; + } + } + + // Read-only info registers + this->mvendorid.write_illegal = true; + this->mimpid.write_illegal = true; + this->marchid.write_illegal = true; + this->mhartid.write_illegal = true; + + // ---------------------------------------------------------------- + // CSR write masks — values from CV32E40P RTL + // Config Python values are used if present; fallback to RTL-derived + // constants if config key returns 0 (missing). + // ---------------------------------------------------------------- + + // mstatus effective write mask. + // CV32E40P forces MPP=M, MPRV=0, UIE=0, UPIE=0. + // Only MIE(3) + MPIE(7) are truly writable. With FPU: add FS(14:13). + // Reset is ALWAYS 0x1800 (FS=Off) + // even when FPU=1. FS transitions to Dirty + // only when FPU registers are written, not at reset. + iss_reg_t mstatus_wmask_dflt = this->m_fpu_in_isa ? 0x6088 : 0x0088; + iss_reg_t mstatus_reset = 0x00001800; // MPP=M, FS=Off for ALL configs + this->mstatus.set_write_mask(cfg_int_or(cfg, "mstatus_write_mask", (int)mstatus_wmask_dflt)); + this->mstatus.reset_val = mstatus_reset; + this->mstatus.value = mstatus_reset; + + // mie — only IRQ_MASK bits writable + this->mie.set_write_mask(cfg_int_or(cfg, "mie_write_mask", 0xFFFF0888)); + + // mip — read-only in M-mode + this->mip.set_write_mask(0); + + // mtvec — bits[7:1] hardwired to 0 + this->mtvec.set_write_mask(cfg_int_or(cfg, "mtvec_write_mask", 0xFFFFFF01)); + this->mtvec.reset_val = cfg_int_or(cfg, "mtvec_reset", 0x1); + this->mtvec.value = this->mtvec.reset_val; // build() runs AFTER reset() + + // mtval — CV32E40P hardwires mtval to 0 (not writable) + this->mtval.set_write_mask(cfg_int_or(cfg, "mtval_write_mask", 0x00000000)); + + // mcause — bit[31] (interrupt) + bits[4:0] (exception code) + this->mcause.set_write_mask(cfg_int_or(cfg, "mcause_mask", 0x8000001F)); + + // ---------------------------------------------------------------- + // Trigger module CSRs + // ---------------------------------------------------------------- + // tselect — CV32E40P has exactly 1 trigger, tselect hardwired to 0 + this->tselect.reset_val = 0; + this->tselect.value = 0; + this->tselect.set_write_mask(0); // read-only: writes ignored, always reads 0 + // tselect reads always return reset_val (0). + this->tselect_default_read = this->tselect.reset_val; + + // tdata1 — type=2 (mcontrol), dmode=1, action=1, match=0, m=1 + this->tdata1.reset_val = cfg_int_or(cfg, "tdata1_reset", 0x28001040); + this->tdata1.value = this->tdata1.reset_val; + // tdata1 is writable ONLY from Debug Mode + // In M-mode, all writes are silently ignored. + // Since GVSOC model doesn't implement Debug Mode, mask=0. + this->tdata1.set_write_mask(cfg_int_or(cfg, "tdata1_write_mask", 0x00000000)); + + // tdata2 writable ONLY from Debug Mode (cv32e40p_cs_registers.sv:1263) + this->tdata2.set_write_mask(cfg_int_or(cfg, "tdata2_write_mask", 0x00000000)); + this->tdata3.set_write_mask(cfg_int_or(cfg, "tdata3_write_mask", 0x00000000)); + + // mcontext/scontext — writable only from Debug Mode + this->mcontext.set_write_mask(cfg_int_or(cfg, "mcontext_write_mask", 0x00000000)); + this->scontext.set_write_mask(cfg_int_or(cfg, "scontext_write_mask", 0x00000000)); + + // tinfo — read-only, bit[2]=1 (type 2 = mcontrol supported) + this->tinfo.reset_val = cfg_int_or(cfg, "tinfo_reset", 0x4); + this->tinfo.value = this->tinfo.reset_val; + this->tinfo.set_write_mask(0); // read-only + + // ---------------------------------------------------------------- + // Read-only info registers — initialization + // ---------------------------------------------------------------- + // mvendorid — OpenHW JEDEC bank 13, RISC-V compliant + this->mvendorid.reset_val = cfg_int_or(cfg, "mvendorid_value", 0x00000602); + this->mvendorid.value = this->mvendorid.reset_val; + + // marchid — CV32E40P architecture ID + this->marchid.reset_val = cfg_int_or(cfg, "marchid_value", 0x00000004); + this->marchid.value = this->marchid.reset_val; + + // mhartid — hart ID (default 0, parameterizable per config) + this->mhartid.reset_val = cfg_int_or(cfg, "mhartid_value", 0x00000000); + this->mhartid.value = this->mhartid.reset_val; + + // mimpid from JSON config (RTL: FPU||COREV_PULP||COREV_CLUSTER ? 1 : 0) + this->mimpid.value = (iss_reg_t)cfg_int_or(cfg, "mimpid", 0); + this->mimpid.reset_val = this->mimpid.value; + + // ---------------------------------------------------------------- + // Counter CSRs + // ---------------------------------------------------------------- + + // mcountinhibit — RTL resets to write_mask value (all implemented bits set). + // With N=1 HPM counter: mask=0x0d (CY,IR,HPM3). With N=29: mask=0xfffffffd. + iss_reg_t mcountinhibit_mask = (iss_reg_t)cfg_int_or(cfg, "mcountinhibit_mask", 0x0d); + this->mcountinhibit.set_write_mask(mcountinhibit_mask); + this->mcountinhibit.reset_val = mcountinhibit_mask; + this->mcountinhibit.value = mcountinhibit_mask; + + // HPM counters: only first num_mhpmcounters are implemented + int num_hpm = cfg_int_or(cfg, "num_mhpmcounters", 1); + for (int i = 0; i < 29; i++) + { + if (i < num_hpm) + { + this->mhpmcounter[i].set_write_mask((iss_reg_t)-1); +#if ISS_REG_WIDTH == 32 + this->mhpmcounterh[i].set_write_mask((iss_reg_t)-1); +#endif + } + else + { + // CV32E40P: unimplemented counters are WARL (writes ignored, no exception) + this->mhpmcounter[i].set_write_mask(0); +#if ISS_REG_WIDTH == 32 + this->mhpmcounterh[i].set_write_mask(0); +#endif + } + } + + // HPM event selectors: only mhpmevent3 has bits[15:0] + int num_hpm_events = cfg_int_or(cfg, "num_hpm_events", 16); + iss_reg_t evt_mask = (num_hpm_events > 0) ? + ((1U << num_hpm_events) - 1) : 0; + for (int i = 0; i < 29; i++) + { + this->mhpmevent[i].set_write_mask((i < num_hpm) ? evt_mask : 0); + } + + // CV32E40P raises illegal-instruction on access to undeclared CSRs + // (RISC-V privileged spec strict mode); generic GVSOC just warns. + this->raise_on_unsupported_csr_flag = true; + // CV32E40P bootaddr does NOT write mtvec — it is set by Csr::build() + // (reset_val=0x1, vectored mode). + this->bootaddr_writes_mtvec_flag = false; + + // dcsr — build() runs after reset(), so apply dcsr M-mode here too + this->dcsr = (4 << 28) | 0x3; +} + +void Cv32e40pCsr::reset(bool active) +{ + Csr::reset(active); + + // dcsr reset — xdebugver=4 in [31:28], prv=M-mode in [1:0] + // Base Csr::reset() sets dcsr = 4 << 28 (prv=0). CV32E40P is M-mode only. + this->dcsr = (4 << 28) | 0x3; + + this->mcycle_offset = 0; + + // Re-apply config-driven values after Csr::reset() which may clobber them + // with base-class defaults (e.g. marchid=0x14 from GVSOC base instead of + // 0x04 from CV32E40P RTL). Uses reset_val set by build_cv32e40p(). + if (active) + { + this->mvendorid.value = this->mvendorid.reset_val; + this->marchid.value = this->marchid.reset_val; + this->mimpid.value = this->mimpid.reset_val; + + // mstatus — Core::build() contaminates reset_val with FS=Dirty. + // CV32E40P: reset is ALWAYS 0x1800 (FS=Off, MPP=M) for ALL configs + // including FPU (RTL: mstatus_fs_q <= FS_OFF at reset). + iss_reg_t mstatus_rst = 0x00001800; + this->mstatus.reset_val = mstatus_rst; + this->mstatus.value = mstatus_rst; + + // tdata1 — re-apply value from build_cv32e40p() after Csr::reset() + // may have used a contaminated reset_val. build_cv32e40p() reads + // tdata1_reset from JSON config (default 0x28001040: m=1, u=0). + this->tdata1.value = this->tdata1.reset_val; + } +} + +bool Cv32e40pCsr::mstatus_access(bool is_write, iss_reg_t &value) +{ + // Return true → let CsrAbtractReg::access() handle the write_mask. + // For reads, force MPP=M so the value read is correct. + if (!is_write) + { + this->mstatus.mpp = PRIV_M; + value = this->mstatus.value; + return false; + } + return true; +} + +bool Cv32e40pCsr::fp_access_illegal() +{ + // CV32E40P RTL (cv32e40p_cs_registers.sv): FP CSR (fflags/frm/fcsr) access + // raises illegal-instruction when mstatus[FS] == 00 (FS_OFF). + return ((this->mstatus.value >> 13) & 3) == 0; +} + +void Cv32e40pCsr::fp_state_dirty() +{ + // CV32E40P RTL (cv32e40p_cs_registers.sv:1027-1037): when FPU=1 && ZFINX=0, + // mstatus.FS is forced to FS_DIRTY(2'b11) on FP regfile write, fflags + // update, or FP-CSR write. Gated on m_fpu_in_isa (false for ZFINX) to + // match fp_access_illegal / SD read-fixup. SD(bit31) is derived on read + // (SD = FS==3) — not stored here. + if (this->m_fpu_in_isa) + { + this->mstatus.value = (this->mstatus.value & ~(0x3 << 13)) | (0x3 << 13); + } +} + +int Cv32e40pCsr::hwloop_csr_index(iss_reg_t reg) +{ + // CoreV2 HWLOOP CSR addresses (gap at 0xCC3): + // 0xCC0..0xCC2 → loop 0 (start/end/count) → indices 0..2 + // 0xCC4..0xCC6 → loop 1 (start/end/count) → indices 4..6 + // Internal hwloop_regs[] uses stride 4 (per loop slot), matching CoreV2. + if ((reg >= 0xCC0 && reg <= 0xCC2) || (reg >= 0xCC4 && reg <= 0xCC6)) + { + return reg - 0xCC0; + } + return -1; +} + +const char *Cv32e40pCsr::custom_csr_name(iss_reg_t reg) +{ + switch (reg) + { + case 0xCC0: return "lpstart0"; + case 0xCC1: return "lpend0"; + case 0xCC2: return "lpcount0"; + case 0xCC4: return "lpstart1"; + case 0xCC5: return "lpend1"; + case 0xCC6: return "lpcount1"; + default: return nullptr; // fall through to base/generic + } +} + +bool Cv32e40pCsr::mcycle_access(bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->mcycle.value = value; + this->mcycle_offset = (int64_t)value + - (int64_t)this->iss.top.clock.get_cycles(); + } + else + { + if (this->mcountinhibit.value & 0x1) + { + value = this->mcycle.value; + } + else + { + value = (iss_reg_t)((int64_t)this->iss.top.clock.get_cycles() + + this->mcycle_offset); + } + } + return false; +} + +void Cv32e40pCsr::mstatus_read_fixup(iss_reg_t &value) +{ + /* CV32E40P: set mstatus.SD (bit 31) when FS[14:13]==3 or XS[16:15]==3. + * This matches read-back behavior for mstatus. */ + if (this->m_fpu_in_isa && (((value >> 13) & 3) == 3 || ((value >> 15) & 3) == 3)) + { + value |= (1ULL << 31); + } +} + +// EBREAK in M-mode enters debug when dcsr.ebreakm=1. +bool Cv32e40pCsr::ebreak_m_mode_enters_debug() +{ + return (this->dcsr >> 15) & 1; +} + +#endif /* CONFIG_GVSOC_ISS_CV32E40P */ diff --git a/models/cpu/iss/src/cv32e40p/irq_cv32e40p.cpp b/models/cpu/iss/src/cv32e40p/irq_cv32e40p.cpp new file mode 100644 index 000000000..e1ee9e727 --- /dev/null +++ b/models/cpu/iss/src/cv32e40p/irq_cv32e40p.cpp @@ -0,0 +1,103 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +#ifdef CONFIG_GVSOC_ISS_CV32E40P + +#include +#include + +void Cv32e40pIrq::register_csr_callbacks() +{ + this->iss.csr.mip.register_callback(std::bind(&Cv32e40pIrq::mip_access, this, std::placeholders::_1, std::placeholders::_2)); + this->iss.csr.mie.register_callback(std::bind(&Cv32e40pIrq::mie_access, this, std::placeholders::_1, std::placeholders::_2)); + this->iss.csr.mtvec.register_callback(std::bind(&Cv32e40pIrq::mtvec_access, this, std::placeholders::_1, std::placeholders::_2)); +} + +bool Cv32e40pIrq::mip_access(bool is_write, iss_reg_t &value) +{ + if (is_write) + { + iss_reg_t mask = this->iss.csr.mip.write_mask; + this->iss.csr.mip.value = (this->iss.csr.mip.value & ~mask) | (value & mask); + } + else + { + value = this->iss.csr.mip.value; + } + this->check_interrupts(); + return false; +} + +bool Cv32e40pIrq::mie_access(bool is_write, iss_reg_t &value) +{ + if (is_write) + { + iss_reg_t mask = this->iss.csr.mie.write_mask; + this->iss.csr.mie.value = (this->iss.csr.mie.value & ~mask) | (value & mask); + } + else + { + value = this->iss.csr.mie.value; + } + this->check_interrupts(); + return false; +} + +bool Cv32e40pIrq::mtvec_access(bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->mtvec_set(value); + iss_reg_t mask = this->iss.csr.mtvec.write_mask; + this->iss.csr.mtvec.value = (this->iss.csr.mtvec.value & ~mask) | (value & mask); + return false; + } + else + { + value = this->iss.csr.mtvec.value; + return true; + } +} + +void Cv32e40pIrq::elw_irq_unstall() +{ + this->trace.msg("Interrupting pending elw\n"); + this->iss.exec.current_insn = this->iss.exec.elw_insn; + this->iss.exec.elw_interrupted = 1; + this->iss.exec.busy_enter(); +} + +/* CV32E40P vectored trap entry. mtvec.value keeps the MODE bit (write_mask + * 0xFFFFFF01): MODE=1 (vectored) -> an interrupt enters at (base & ~1) + cause*4; + * direct mode (and exceptions) enter at the base. Matches the RTL mtvec.MODE. */ +iss_reg_t Cv32e40pIrq::compute_trap_entry(iss_reg_t base, int cause, bool is_interrupt) +{ + iss_reg_t vbase = base & ~(iss_reg_t)1; + if (is_interrupt && (base & 1)) + { + return vbase + (iss_reg_t)cause * 4; + } + return vbase; +} + + +#endif /* CONFIG_GVSOC_ISS_CV32E40P */ diff --git a/models/cpu/iss/src/exception.cpp b/models/cpu/iss/src/exception.cpp index e1a697486..2387caa2c 100644 --- a/models/cpu/iss/src/exception.cpp +++ b/models/cpu/iss/src/exception.cpp @@ -80,12 +80,20 @@ void Exception::raise(iss_reg_t pc, int id) #ifdef CONFIG_GVSOC_ISS_RISCV_EXCEPTIONS if (next_mode == PRIV_M) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + pc = this->iss.csr.mtvec.value & this->trap_vector_align_mask; +#else pc = this->iss.csr.mtvec.value; +#endif this->iss.csr.mcause.value = id; } else { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + pc = this->iss.csr.stvec.value & this->trap_vector_align_mask; +#else pc = this->iss.csr.stvec.value; +#endif this->iss.csr.scause.value = id; } #else diff --git a/models/cpu/iss/src/exec/exec_inorder.cpp b/models/cpu/iss/src/exec/exec_inorder.cpp index 348deff52..b8c34a6d3 100644 --- a/models/cpu/iss/src/exec/exec_inorder.cpp +++ b/models/cpu/iss/src/exec/exec_inorder.cpp @@ -170,6 +170,15 @@ void Exec::exec_instr(vp::Block *__this, vp::ClockEvent *event) if (iss->exec.handle_stall_cycles()) return; +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Combinatorial DECODE-stage IRQ check: redirect to mtvec before fetching + // the next instruction if an interrupt is pending. + if (!iss->exec.skip_irq_check && iss->irq.check()) + { + return; + } + +#endif iss->exec.trace.msg(vp::Trace::LEVEL_TRACE, "Handling instruction with fast handler\n"); iss_reg_t pc = iss->exec.current_insn; @@ -327,7 +336,17 @@ void Exec::exec_instr_check_all(vp::Block *__this, vp::ClockEvent *event) if (!_this->skip_irq_check) { +#ifdef CONFIG_GVSOC_ISS_CV32E40P + /* Combinatorial DECODE-stage IRQ check (mirrors RTL controller): on a + * pending interrupt, redirect to mtvec before fetching next insn so + * mepc captures the right PC. */ + if (_this->iss.irq.check()) + { + return; + } +#else _this->iss.irq.check(); +#endif } else { @@ -434,8 +453,18 @@ void Exec::fetchen_sync(vp::Block *__this, bool active) void Exec::bootaddr_apply(uint32_t value) { this->trace.msg("Setting boot address (value: 0x%x)\n", value); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // Config flag: CV32E40P keeps its own mtvec reset value instead of + // deriving mtvec from the boot address. + if (this->iss.csr.bootaddr_writes_mtvec_flag) + { + iss_reg_t bootaddr = this->bootaddr_reg.get() & ~((1 << 8) - 1); + this->iss.csr.mtvec.access(true, bootaddr); + } +#else iss_reg_t bootaddr = this->bootaddr_reg.get() & ~((1 << 8) - 1); this->iss.csr.mtvec.access(true, bootaddr); +#endif } diff --git a/models/cpu/iss/src/irq/irq_riscv.cpp b/models/cpu/iss/src/irq/irq_riscv.cpp index 6bea2afb3..417d131bd 100644 --- a/models/cpu/iss/src/irq/irq_riscv.cpp +++ b/models/cpu/iss/src/irq/irq_riscv.cpp @@ -33,12 +33,19 @@ void Irq::build() iss.top.traces.new_trace("irq", &this->trace, vp::DEBUG); this->iss.csr.mideleg.register_callback(std::bind(&Irq::mideleg_access, this, std::placeholders::_1, std::placeholders::_2)); +#ifndef CONFIG_GVSOC_ISS_CV32E40P this->iss.csr.mip.register_callback(std::bind(&Irq::mip_access, this, std::placeholders::_1, std::placeholders::_2)); this->iss.csr.mie.register_callback(std::bind(&Irq::mie_access, this, std::placeholders::_1, std::placeholders::_2)); +#endif this->iss.csr.sip.register_callback(std::bind(&Irq::sip_access, this, std::placeholders::_1, std::placeholders::_2)); this->iss.csr.sie.register_callback(std::bind(&Irq::sie_access, this, std::placeholders::_1, std::placeholders::_2)); +#ifndef CONFIG_GVSOC_ISS_CV32E40P this->iss.csr.mtvec.register_callback(std::bind(&Irq::mtvec_access, this, std::placeholders::_1, std::placeholders::_2)); +#endif this->iss.csr.stvec.register_callback(std::bind(&Irq::stvec_access, this, std::placeholders::_1, std::placeholders::_2)); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + this->register_csr_callbacks(); +#endif this->msi_itf.set_sync_meth(&Irq::msi_sync); this->iss.top.new_slave_port("msi", &this->msi_itf, (vp::Block *)this); @@ -60,6 +67,15 @@ void Irq::build() } } +#ifdef CONFIG_GVSOC_ISS_CV32E40P +void Irq::register_csr_callbacks() +{ + this->iss.csr.mip.register_callback(std::bind(&Irq::mip_access, this, std::placeholders::_1, std::placeholders::_2)); + this->iss.csr.mie.register_callback(std::bind(&Irq::mie_access, this, std::placeholders::_1, std::placeholders::_2)); + this->iss.csr.mtvec.register_callback(std::bind(&Irq::mtvec_access, this, std::placeholders::_1, std::placeholders::_2)); +} + +#endif void Irq::reset(bool active) { if (active) @@ -71,8 +87,10 @@ void Irq::reset(bool active) } else { +#ifndef CONFIG_GVSOC_ISS_CV32E40P this->mtvec_set(this->iss.exec.bootaddr_reg.get() & ~((1 << 8) - 1)); this->stvec_set(this->iss.exec.bootaddr_reg.get() & ~((1 << 8) - 1)); +#endif } } @@ -216,6 +234,13 @@ bool Irq::stvec_set(iss_addr_t base) return true; } +#ifdef CONFIG_GVSOC_ISS_CV32E40P +void Irq::elw_irq_unstall() +{ + // Base implementation: no-op. The CV32E40P subclass overrides this. +} + +#endif void Irq::cache_flush() { } @@ -269,6 +294,14 @@ void Irq::check_interrupts() } } +#ifdef CONFIG_GVSOC_ISS_CV32E40P +// CV32E40P: default trap-entry returns the base unchanged (upstream direct-mode +// behaviour). Cv32e40pIrq overrides this to implement vectored mode. +iss_reg_t Irq::compute_trap_entry(iss_reg_t base, int /*cause*/, bool /*is_interrupt*/) +{ + return base; +} +#endif int Irq::check() { if (this->req_debug && !this->iss.exec.debug_mode) @@ -353,7 +386,14 @@ int Irq::check() this->iss.csr.mstatus.mie = 0; this->iss.csr.mstatus.mpie = this->iss.irq.irq_enable.get(); this->iss.csr.mstatus.mpp = this->iss.core.mode_get(); +#ifdef CONFIG_GVSOC_ISS_CV32E40P + // CV32E40P: honour mtvec.MODE (vectored -> base + cause*4). The + // default Irq::compute_trap_entry() returns the base unchanged, so + // non-CV32E40P targets keep the upstream direct-mode behaviour. + this->iss.exec.current_insn = this->compute_trap_entry(this->iss.csr.mtvec.value, irq, true); +#else this->iss.exec.current_insn = this->iss.csr.mtvec.value; +#endif this->iss.csr.mcause.value = (1ULL << (ISS_REG_WIDTH - 1)) | (unsigned int)irq; } else diff --git a/models/cpu/iss/src/lsu.cpp b/models/cpu/iss/src/lsu.cpp index c7b710665..d6109b4b3 100644 --- a/models/cpu/iss/src/lsu.cpp +++ b/models/cpu/iss/src/lsu.cpp @@ -555,7 +555,11 @@ bool Lsu::atomic(iss_insn_t *insn, iss_addr_t addr, int size, int reg_in, int re // uint8_t *check_second_data = (uint8_t *)this->iss.regfile.reg_ref(reg_out); // req->set_second_memcheck_data(check_second_data); // #endif +#ifdef CONFIG_GVSOC_ISS_CV32E40P + req->set_initiator(this->iss.csr.mhartid.value); +#else req->set_initiator(this->iss.csr.mhartid); +#endif this->log_addr.set_and_release(addr); this->log_size.set_and_release(size); diff --git a/models/cpu/iss_v2/include/csr.hpp b/models/cpu/iss_v2/include/csr.hpp index 77bf1db69..852ac49f2 100644 --- a/models/cpu/iss_v2/include/csr.hpp +++ b/models/cpu/iss_v2/include/csr.hpp @@ -54,6 +54,11 @@ class CsrAbtractReg iss_reg_t reset_val; bool write_illegal = false; + // Public setter so core-specific Csr subclasses can tighten the write + // mask of registers declared by the base class (declare_csr sets it only + // at declaration time). + void set_write_mask(iss_reg_t mask) { this->write_mask = mask; } + protected: void reset(bool active); @@ -160,6 +165,12 @@ class Csr bool access(iss_insn_t *insn, bool is_write, iss_reg_t address, iss_reg_t &value); + // When set, accessing a CSR that no path handles raises an + // illegal-instruction exception instead of only logging a warning. + // Cores that follow the privileged spec strictly (e.g. CV32E40P) + // enable this from their Csr subclass constructor. + bool raise_on_unsupported_csr = false; + Iss &iss; vp::Trace trace; @@ -258,6 +269,13 @@ class Csr #endif #endif +protected: + // Remove a CSR declared by the base class. Core-specific subclasses use + // it to drop registers their core does not implement (e.g. the S-mode + // set on an M-mode-only core), so access falls through to the + // unsupported-CSR path. + void undeclare_csr(iss_reg_t address) { this->regs.erase(address); } + private: bool tselect_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); diff --git a/models/cpu/iss_v2/include/hwloop/hwloop.hpp b/models/cpu/iss_v2/include/hwloop/hwloop.hpp index 4691269df..13584c954 100644 --- a/models/cpu/iss_v2/include/hwloop/hwloop.hpp +++ b/models/cpu/iss_v2/include/hwloop/hwloop.hpp @@ -50,10 +50,37 @@ class Hwloop // otherwise returns next_pc unchanged. inline iss_reg_t check(iss_reg_t pc, iss_reg_t next_pc); - // Setters (called from ISA decoders and CSR writes). - void set_start(int idx, iss_reg_t pc); - void set_end(int idx, iss_reg_t pc); - void set_count(int idx, iss_reg_t count); + // Setters (called from ISA decoders and CSR writes). Header-inline on + // purpose: external controllers living in a separate shared object can + // only call what inlines (the model library does not export symbols). + void set_start(int idx, iss_reg_t pc) + { + this->trace.msg(vp::Trace::LEVEL_DEBUG, + "Setting hwloop start (idx: %d, pc: 0x%lx)\n", idx, (unsigned long)pc); + this->start_pc[idx] = pc; + } + + void set_end(int idx, iss_reg_t pc) + { + this->trace.msg(vp::Trace::LEVEL_DEBUG, + "Setting hwloop end (idx: %d, pc: 0x%lx)\n", idx, (unsigned long)pc); + this->end_pc[idx] = pc; + } + + void set_count(int idx, iss_reg_t count) + { + this->trace.msg(vp::Trace::LEVEL_DEBUG, + "Setting hwloop count (idx: %d, count: %d)\n", idx, (int)count); + this->count[idx] = count; + if (count == 0) + { + this->active &= ~(1u << idx); + } + else + { + this->active |= (1u << idx); + } + } // Getters (called from CSR reads). iss_reg_t get_start(int idx) const { return this->start_pc[idx]; } diff --git a/models/cpu/iss_v2/include/insn_cache.hpp b/models/cpu/iss_v2/include/insn_cache.hpp index 360c8b08a..98b2a12f4 100644 --- a/models/cpu/iss_v2/include/insn_cache.hpp +++ b/models/cpu/iss_v2/include/insn_cache.hpp @@ -48,6 +48,16 @@ class InsnCache inline void insn_init(iss_insn_t *insn, iss_addr_t addr); InsnPage *page_get(iss_reg_t paddr); + // True when addr falls inside an already-decoded page - the only case + // where a hart store can stale decoded instructions (self-modifying + // code executed without fence.i: cores fetching straight from memory, + // like CV32E40P, see the new code on the next fetch, so the decoded + // cache must stay coherent with the hart's own stores). + inline bool covers(iss_reg_t addr) + { + return this->pages.find(addr >> INSN_PAGE_BITS) != this->pages.end(); + } + // Bumped on every flush (full or mode flush). Consumers caching // pointers into the pages (e.g. the DBT translation cache) compare // it to detect that their cached state went stale. diff --git a/models/cpu/iss_v2/include/irq/irq_riscv.hpp b/models/cpu/iss_v2/include/irq/irq_riscv.hpp index fb4912ce3..11d4f21f0 100644 --- a/models/cpu/iss_v2/include/irq/irq_riscv.hpp +++ b/models/cpu/iss_v2/include/irq/irq_riscv.hpp @@ -23,6 +23,7 @@ #include #include +#include #define IRQ_U_SOFT 0 #define IRQ_S_SOFT 1 diff --git a/models/cpu/iss_v2/include/isa_lib/macros.h b/models/cpu/iss_v2/include/isa_lib/macros.h index 8feea4ccc..42aa94809 100644 --- a/models/cpu/iss_v2/include/isa_lib/macros.h +++ b/models/cpu/iss_v2/include/isa_lib/macros.h @@ -64,8 +64,17 @@ #else #define FREG_GET(reg) (iss->regfile.get_freg(insn->in_regs[reg])) #define FREG_OUT_GET(reg) (iss->regfile.get_freg(insn->out_regs[reg])) +#if defined(CONFIG_GVSOC_ISS_FP_STATE_DIRTY) +// Opt-in for cores whose csr class tracks mstatus.FS on FP write-backs. +#define FREG_SET(reg,val) (iss->regfile.set_freg(insn->out_regs[reg], val), iss->csr.fp_state_dirty()) +#else #define FREG_SET(reg,val) (iss->regfile.set_freg(insn->out_regs[reg], val)) #endif +#endif +#if defined(CONFIG_GVSOC_ISS_FP_STATE_DIRTY) +#define FREG32_SET(reg,val) (iss->regfile.set_freg(insn->out_regs[reg], val), iss->csr.fp_state_dirty()) +#else #define FREG32_SET(reg,val) (iss->regfile.set_freg(insn->out_regs[reg], val)) +#endif #endif diff --git a/models/cpu/iss_v2/include/lsu_v2.hpp b/models/cpu/iss_v2/include/lsu_v2.hpp index 61a65503e..c8b990329 100644 --- a/models/cpu/iss_v2/include/lsu_v2.hpp +++ b/models/cpu/iss_v2/include/lsu_v2.hpp @@ -177,9 +177,15 @@ class LsuV2 bool handle_req_response(LsuReqEntry *entry); void handle_req_end(LsuReqEntry *entry); // Re-arm and re-issue ``entry`` for the second beat of a misaligned - // access. Returns true if a second beat was fired (entry still in - // flight); false if the entry was aligned (no second beat needed). + // access. Returns true while the completion is still in flight + // (task, granted or denied); false when beat 1 completed inline with + // zero latency — handle_req_end has run and the entry is back on the + // free list, so a caller holding the insn must retire it itself. bool fire_misaligned_second(LsuReqEntry *entry); + // Terminate a held insn the way the normal response path does: + // delayed scoreboard release when a scoreboard is wired, immediate + // otherwise. Used when a misaligned beat 1 completes inline. + void retire_held_insn(InsnEntry *insn_entry); protected: Iss &iss; diff --git a/models/cpu/iss_v2/include/prefetch/prefetch_single_line.hpp b/models/cpu/iss_v2/include/prefetch/prefetch_single_line.hpp index aa44360f2..dde24a03e 100644 --- a/models/cpu/iss_v2/include/prefetch/prefetch_single_line.hpp +++ b/models/cpu/iss_v2/include/prefetch/prefetch_single_line.hpp @@ -91,6 +91,14 @@ class PrefetchSingleLine // Start address of the prefetch buffer. Can be -1 to indicate it is empty iss_addr_t buffer_start_addr; + // Explicit "buffer holds valid data" flag. The buffer_start_addr=-1 empty + // sentinel alone is NOT enough: the fast-path index = addr - buffer_start_addr + // is unsigned, so it wraps to addr+1 for every address, and in the first + // bytes of the address space (addr <= 0x0b with a 16-byte line) that index + // is still in-window and the stale line is read without a refill. This flag + // forces a real refill after any flush() regardless of the address. + bool buffer_valid = false; + // Request used for sending fetch request to the fetch interface vp::IoReq fetch_req; diff --git a/models/cpu/iss_v2/riscv.py b/models/cpu/iss_v2/riscv.py index 1c0d77063..2fcaa653d 100644 --- a/models/cpu/iss_v2/riscv.py +++ b/models/cpu/iss_v2/riscv.py @@ -305,6 +305,9 @@ class RiscvCommon(st.Component): A list of path to riscv binaries (default: []). debug_handler : int, optional The address where the core should jump when switching to debug mode (default: 0). + debug_exception_handler : int, optional + The address where the core should jump when an exception is taken while + in debug mode (dm_exception_addr). Defaults to debug_handler. power_models : dict, optional A dictionnay describing all the power models used to estimate power consumption in the ISS (default: {}) power_models_file : file, optional @@ -323,6 +326,7 @@ def __init__(self, riscv_dbg_unit: bool=False, binaries: list[str]=[], debug_handler: int=0, + debug_exception_handler: int|None=None, power_models: dict[str,Any]={}, power_models_file: str=None, cluster_id: int=0, @@ -532,6 +536,8 @@ def __init__(self, 'riscv_dbg_unit': riscv_dbg_unit, 'binaries': binaries.copy(), 'debug_handler': debug_handler, + 'debug_exception_handler': + debug_exception_handler if debug_exception_handler is not None else debug_handler, 'power_models': power_models, 'cluster_id': cluster_id, 'core_id': config.hart_id, diff --git a/models/cpu/iss_v2/src/core.cpp b/models/cpu/iss_v2/src/core.cpp index f694df699..2b3bb1601 100644 --- a/models/cpu/iss_v2/src/core.cpp +++ b/models/cpu/iss_v2/src/core.cpp @@ -25,6 +25,13 @@ Core::Core(Iss &iss) this->iss.traces.new_trace("core", &this->trace, vp::DEBUG); // Initialize the mstatus write mask so that WPRI fields are preserved +#if defined(CONFIG_GVSOC_ISS_CORE_MSTATUS_WRITE_MASK) + // The core recipe owns the whole mstatus policy: its mask applies as-is + // and none of the generic refinements below (vector/user state, forced + // dirty FS/SD in the reset value) is wanted. Typical user: an M-mode-only + // core whose RTL hardwires most fields and controls FS itself. + this->mstatus_write_mask = CONFIG_GVSOC_ISS_CORE_MSTATUS_WRITE_MASK; +#else #if ISS_REG_WIDTH == 64 this->mstatus_write_mask = 0x8000003F007FFFEA; #else @@ -49,6 +56,7 @@ Core::Core(Iss &iss) this->mstatus_write_mask &= ~(0x3ULL << 34); this->iss.csr.mstatus.reset_val |= 2ULL << 34; #endif +#endif #if ISS_REG_WIDTH == 64 this->sstatus_write_mask = this->mstatus_write_mask & 0x80000003000DE762; diff --git a/models/cpu/iss_v2/src/csr.cpp b/models/cpu/iss_v2/src/csr.cpp index b2c50ac77..49fdccf3d 100644 --- a/models/cpu/iss_v2/src/csr.cpp +++ b/models/cpu/iss_v2/src/csr.cpp @@ -806,6 +806,22 @@ bool iss_csr_read(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t *value) return false; } + // Strict cores treat the declared-register map as the whole CSR space: + // anything it does not contain raises illegal-instruction, without + // falling back to the legacy dispatch below. + if (iss->csr.raise_on_unsupported_csr) + { + // The instruction handlers issue the read and the write of a csrrw + // back-to-back; raise only once as Exception::raise is not idempotent + // (the second call would capture mpie after irq_enable was cleared). + if (!iss->exec.has_exception) + { + iss->csr.trace.msg(vp::Trace::LEVEL_DEBUG, "Unsupported CSR read (id: 0x%x) -> illegal instruction\n", reg); + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + } + return true; + } + // And dispatch switch (reg) { @@ -1190,6 +1206,20 @@ bool iss_csr_write(Iss *iss, iss_insn_t *insn, iss_reg_t reg, iss_reg_t value) return false; } + // Strict cores: same rule as the read path, the declared-register map + // is the whole CSR space. + if (iss->csr.raise_on_unsupported_csr) + { + // See the matching block in iss_csr_read: raise at most once per + // instruction, the csrrw handlers call both functions unconditionally. + if (!iss->exec.has_exception) + { + iss->csr.trace.msg(vp::Trace::LEVEL_DEBUG, "Unsupported CSR write (id: 0x%x) -> illegal instruction\n", reg); + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + } + return true; + } + // And dispatch switch (reg) { diff --git a/models/cpu/iss_v2/src/exec/exec_inorder.cpp b/models/cpu/iss_v2/src/exec/exec_inorder.cpp index 10e71d2ce..cebaab38e 100644 --- a/models/cpu/iss_v2/src/exec/exec_inorder.cpp +++ b/models/cpu/iss_v2/src/exec/exec_inorder.cpp @@ -168,6 +168,21 @@ void ExecInOrder::exec_instr(vp::Block *__this, vp::ClockEvent *event) iss->exec.trace.msg(vp::Trace::LEVEL_TRACE, "Handling instruction with fast handler\n"); + // Honor a pending front-end flush in the FAST path too (the slow handler + // already does at exec_instr_check_all): after an external redirect + // (gvsoc_engine_set_pc sets pending_flush) the very next dispatch can be the + // fast handler, which otherwise fetches the redirected PC through the STALE + // prefetch line (PrefetchSingleLine::fetch's in-window fast-path reads + // this->data[index] without a refill) and decodes garbage bytes -> a bogus + // next-PC at the vector entry. Flushing the prefetch (buffer_start_addr=-1) + // and insn cache here forces a real refill+redecode at current_insn. + if (unlikely(iss->exec.pending_flush)) + { + iss->prefetch.flush(); + iss->insn_cache.flush(); + iss->exec.pending_flush = false; + } + // Leave now in case the core is retained and we are only executing tasks if (unlikely(iss->exec.handle_tasks())) return; @@ -208,7 +223,13 @@ void ExecInOrder::exec_instr(vp::Block *__this, vp::ClockEvent *event) // Hardware-loop redirect: if pc matches a registered loop end // and its counter > 0, decrement and redirect to the loop start. // The default HwloopEmpty variant inlines to a no-op. - next_pc = iss->hwloop.check(pc, next_pc); + // A trapping loop-end instruction is killed before the loop + // update on the RTL (the exception wins), so the count must not + // move when the handler raised. + if (likely(!iss->exec.has_exception)) + { + next_pc = iss->hwloop.check(pc, next_pc); + } iss->exec.current_insn = next_pc; @@ -282,14 +303,11 @@ void ExecInOrder::exec_instr_check_all(vp::Block *__this, vp::ClockEvent *event) _this->switch_to_fast_mode(); } - if (!_this->skip_irq_check) - { - _this->iss.irq.check(); - } - else - { - _this->skip_irq_check = false; - } + // The one-shot async gate (skip_irq_check) is consumed inside check(): + // synchronous debug conditions (execute-address triggers) are evaluated + // on every boundary, even when interrupt checking is suppressed for the + // dispatch (external lockstep stepping, gdb resume). + _this->iss.irq.check(); // Leave now in case the core is retained and we are only executing tasks if (_this->handle_tasks()) return; @@ -345,8 +363,12 @@ void ExecInOrder::exec_instr_check_all(vp::Block *__this, vp::ClockEvent *event) return; } - // Hardware-loop redirect: see fast-path equivalent above. - next_pc = iss->hwloop.check(pc, next_pc); + // Hardware-loop redirect: see fast-path equivalent above (including + // the trapping-insn carve-out). + if (likely(!_this->has_exception)) + { + next_pc = iss->hwloop.check(pc, next_pc); + } _this->current_insn = next_pc; diff --git a/models/cpu/iss_v2/src/hwloop/hwloop.cpp b/models/cpu/iss_v2/src/hwloop/hwloop.cpp index ae4f4cc5a..6d65e9fe3 100644 --- a/models/cpu/iss_v2/src/hwloop/hwloop.cpp +++ b/models/cpu/iss_v2/src/hwloop/hwloop.cpp @@ -35,35 +35,3 @@ void Hwloop::reset(bool active) this->active = 0; } } - - -void Hwloop::set_start(int idx, iss_reg_t pc) -{ - this->trace.msg(vp::Trace::LEVEL_DEBUG, - "Setting hwloop start (idx: %d, pc: 0x%lx)\n", idx, (unsigned long)pc); - this->start_pc[idx] = pc; -} - - -void Hwloop::set_end(int idx, iss_reg_t pc) -{ - this->trace.msg(vp::Trace::LEVEL_DEBUG, - "Setting hwloop end (idx: %d, pc: 0x%lx)\n", idx, (unsigned long)pc); - this->end_pc[idx] = pc; -} - - -void Hwloop::set_count(int idx, iss_reg_t count) -{ - this->trace.msg(vp::Trace::LEVEL_DEBUG, - "Setting hwloop count (idx: %d, count: %d)\n", idx, (int)count); - this->count[idx] = count; - if (count == 0) - { - this->active &= ~(1u << idx); - } - else - { - this->active |= (1u << idx); - } -} diff --git a/models/cpu/iss_v2/src/irq/irq_external.cpp b/models/cpu/iss_v2/src/irq/irq_external.cpp index 1f812ff74..bebd56851 100644 --- a/models/cpu/iss_v2/src/irq/irq_external.cpp +++ b/models/cpu/iss_v2/src/irq/irq_external.cpp @@ -153,6 +153,17 @@ void IrqExternal::irq_req_sync(vp::Block *__this, int irq) int IrqExternal::check() { + /* One-shot async gate: a dispatch stepped with skip_irq_check set (external + * lockstep stepping, gdb resume) takes no interrupt and serves no + * asynchronous debug request. Consumed here rather than at the call + * site so a personality can evaluate synchronous debug conditions + * (execute-address triggers) ahead of this point on every boundary. */ + if (this->iss.exec.skip_irq_check) + { + this->iss.exec.skip_irq_check = false; + return 0; + } + if (this->req_debug && !this->iss.exec.debug_mode) { this->iss.exec.debug_mode = true; diff --git a/models/cpu/iss_v2/src/irq/irq_riscv.cpp b/models/cpu/iss_v2/src/irq/irq_riscv.cpp index f640fd116..e72bf4f7a 100644 --- a/models/cpu/iss_v2/src/irq/irq_riscv.cpp +++ b/models/cpu/iss_v2/src/irq/irq_riscv.cpp @@ -256,6 +256,17 @@ void IrqRiscv::check_interrupts() int IrqRiscv::check() { + /* One-shot async gate: a dispatch stepped with skip_irq_check set (external + * lockstep stepping, gdb resume) takes no interrupt and serves no + * asynchronous debug request. Consumed here rather than at the call + * site so a personality can evaluate synchronous debug conditions + * (execute-address triggers) ahead of this point on every boundary. */ + if (this->iss.exec.skip_irq_check) + { + this->iss.exec.skip_irq_check = false; + return 0; + } + if (this->req_debug && !this->iss.exec.debug_mode) { this->iss.exec.debug_mode = true; diff --git a/models/cpu/iss_v2/src/lsu_v2.cpp b/models/cpu/iss_v2/src/lsu_v2.cpp index 241c3bedf..451ae472c 100644 --- a/models/cpu/iss_v2/src/lsu_v2.cpp +++ b/models/cpu/iss_v2/src/lsu_v2.cpp @@ -87,6 +87,28 @@ bool LsuV2::data_req_virtual(iss_insn_t *insn, iss_addr_t addr, int size, if (this->iss.mmu.load_virt_to_phys(addr, phys_addr, use_mem_array)) return false; } + /* Self-modifying code: a store into an already-decoded page stales its + * decoded instructions. Queue a flush (deferred to the next dispatch + * boundary by icache_flush - flushing here would free the page holding + * the store insn itself). Rare event, full flush is fine. + * Placed AFTER translation: insn-cache pages are keyed on physical + * addresses (InsnCache::page_get), so the virtual address would be the + * wrong key on an MMU-enabled core. The second lookup only runs when + * the access actually crosses a page (one hash per store on the hot + * path). Known latent gap, accepted: the LsuV2::atomic opcodes (SC/AMO) + * also write memory but no iss_v2 core wires them today - extend the + * opcode test when one does. */ + if (opcode == vp::IoReqOpcode::WRITE && size > 0) + { + iss_addr_t last = phys_addr + (iss_addr_t)size - 1; + if (this->iss.insn_cache.covers(phys_addr) || + (((phys_addr ^ last) >> INSN_PAGE_BITS) != 0 && + this->iss.insn_cache.covers(last))) + { + this->iss.exec.icache_flush(); + } + } + if (this->io_req_denied || this->data_req(insn, addr, size, opcode, is_signed, reg, reg2)) { this->iss.exec.insn_stall(); @@ -158,10 +180,20 @@ vp::IoRespAck LsuV2::data_response(vp::Block *__this, vp::IoReq *req) // First beat of a misaligned access just landed: fire beat 1 and // keep the insn held. Do NOT touch next_retire_cycle yet — that is - // reserved for the *final* retire when beat 1 lands. + // reserved for the *final* retire when beat 1 lands. When beat 1 + // completes inline with zero latency, fire_misaligned_second has + // already run handle_req_end and freed the entry, but the insn is + // held here — retire it now or it stays parked forever. if (entry->misaligned_size != 0) { - _this->fire_misaligned_second(entry); + InsnEntry *held = entry->insn_entry; + if (!_this->fire_misaligned_second(entry)) + { + _this->retire_held_insn(held); + int64_t now = _this->iss.clock.get_cycles(); + if (_this->next_retire_cycle < now + 1) + _this->next_retire_cycle = now + 1; + } return vp::IO_RESP_ACCEPTED; } @@ -477,21 +509,28 @@ bool LsuV2::data_req_misaligned(iss_insn_t *insn, iss_addr_t addr, int size, // data2 so fire_misaligned_second can recover the high bytes. // Peek the entry that data_req_aligned is about to allocate (head of - // the free list). After the call returns, this same entry is the - // in-flight one — `data_req_aligned` keeps it in flight on the - // sync-DONE / GRANTED / DENIED-held paths, all of which return - // ``false`` (no stall). Only the LSU-full path (``true``) leaves no - // entry allocated, so the misaligned bookkeeping is committed on - // ``false``. + // the free list) and arm the beat-1 bookkeeping BEFORE issuing beat 0: + // a beat 0 that completes synchronously with zero latency (e.g. the + // background sparse memory) retires and frees the entry inside the + // call, so arming afterwards would poison the free list and hijack + // the next access allocating this entry. Armed up front, every + // completion path — including the synchronous zero-latency one in + // handle_req_response — sees ``misaligned_size != 0`` and routes + // through fire_misaligned_second instead of retiring after beat 0. LsuReqEntry *entry = this->req_entry_first; - // For writes, snapshot the full register value into data2 BEFORE - // data_req_aligned overwrites entry->data with the truncated beat-0 - // payload. fire_misaligned_second will shift this down for beat 1. - uint64_t full_write_data = 0; - if (opcode == vp::IoReqOpcode::WRITE && entry != nullptr) + if (entry != nullptr) { - full_write_data = this->iss.regfile.get_reg(reg); + entry->misaligned_size = size1; + entry->misaligned_addr = addr1; + entry->misaligned_byte_offset = size0; + if (opcode == vp::IoReqOpcode::WRITE) + { + // Snapshot the full register value into data2 BEFORE + // data_req_aligned overwrites entry->data with the beat-0 + // payload. fire_misaligned_second shifts it down for beat 1. + entry->data2 = this->iss.regfile.get_reg(reg); + } } this->issuing_misaligned = true; @@ -499,20 +538,40 @@ bool LsuV2::data_req_misaligned(iss_insn_t *insn, iss_addr_t addr, int size, is_signed, reg, reg2); this->issuing_misaligned = false; - if (!stalled && entry != nullptr) + if (stalled && entry != nullptr) { - entry->misaligned_size = size1; - entry->misaligned_addr = addr1; - entry->misaligned_byte_offset = size0; - if (opcode == vp::IoReqOpcode::WRITE) - { - entry->data2 = full_write_data; - } + // Beat 0 was not issued (LSU full): the peeked entry is still on + // the free list, disarm it so a future aligned access starts clean. + // Unreachable today (data_req_aligned pops the same head we peeked, + // so entry != NULL implies it was allocated); kept as defensive + // symmetry in case the peek/pop pairing ever changes. + entry->misaligned_size = 0; + entry->misaligned_byte_offset = 0; } return stalled; } +void LsuV2::retire_held_insn(InsnEntry *insn_entry) +{ + // NOTE: unlike the normal response path this does not serialize on + // next_retire_cycle (the entry that would carry the deferral task is + // already freed by handle_req_end when we get here). Safe while + // nb_outstanding == 1 - a second in-flight retire cannot exist in the + // same cycle - but a config with nb_outstanding > 1 needs a deferral + // mechanism independent of the LsuReqEntry before using this path. +#ifdef CONFIG_GVSOC_ISS_REGFILE_SCOREBOARD + // Same load-use 1-cycle stall as the normal response path: park the + // dest regs' release for next cycle and keep insn_terminate away + // from the scoreboard. + iss_insn_t *insn = this->iss.exec.get_insn(insn_entry); + this->iss.exec.schedule_scoreboard_release(insn->sb_out_reg_mask); + this->iss.exec.insn_terminate(insn_entry, /*defer_scoreboard_release=*/true); +#else + this->iss.exec.insn_terminate(insn_entry); +#endif +} + bool LsuV2::fire_misaligned_second(LsuReqEntry *entry) { int size0 = entry->misaligned_byte_offset; @@ -602,10 +661,18 @@ void LsuV2::task_handle(Iss *iss, Task *task) } // First beat of a misaligned access just timed out: fire beat 1. - // No retire-slot accounting yet — beat 1 will do its own. + // No retire-slot accounting yet — beat 1 will do its own, except + // when it completes inline with zero latency: then the entry is + // already freed and the held insn must be retired here. if (entry->misaligned_size != 0) { - iss->lsu.fire_misaligned_second(entry); + InsnEntry *held = entry->insn_entry; + if (!iss->lsu.fire_misaligned_second(entry)) + { + iss->lsu.retire_held_insn(held); + if (iss->lsu.next_retire_cycle < cur + 1) + iss->lsu.next_retire_cycle = cur + 1; + } return; } diff --git a/models/cpu/iss_v2/src/prefetch/prefetch_single_line.cpp b/models/cpu/iss_v2/src/prefetch/prefetch_single_line.cpp index 968141fb2..17d860460 100644 --- a/models/cpu/iss_v2/src/prefetch/prefetch_single_line.cpp +++ b/models/cpu/iss_v2/src/prefetch/prefetch_single_line.cpp @@ -212,6 +212,7 @@ int PrefetchSingleLine::fill(iss_addr_t addr) { iss_addr_t aligned_addr = addr & ~(CONFIG_GVSOC_ISS_PREFETCH_SIZE - 1); this->buffer_start_addr = aligned_addr; + this->buffer_valid = true; return this->send_fetch_req(aligned_addr, this->data, CONFIG_GVSOC_ISS_PREFETCH_SIZE, false); } @@ -265,24 +266,39 @@ bool PrefetchSingleLine::fetch(iss_reg_t addr) unsigned int index = phys_addr - this->buffer_start_addr; - // If it is entirely within the buffer, get the opcode and decode it. - if (likely(index <= CONFIG_GVSOC_ISS_PREFETCH_SIZE - sizeof(iss_opcode_t))) + // If the buffer is valid and the instruction is entirely within it, get the + // opcode and decode it. buffer_valid guards the flushed/empty state (see + // flush(): the unsigned index would otherwise false-hit a stale low-address + // line). + if (likely(this->buffer_valid && + index <= CONFIG_GVSOC_ISS_PREFETCH_SIZE - sizeof(iss_opcode_t))) { insn->opcode = *(iss_opcode_t *)&this->data[index]; return true; } - // Otherwise, fake a refill + // Miss or invalidated buffer: refill from this address. Force fetch_refill's + // fill() branch when the buffer was flushed, since the wrapped index could + // still look in-range. this->current_pc = addr; + if (!this->buffer_valid) + index = CONFIG_GVSOC_ISS_PREFETCH_SIZE; return this->fetch_refill(insn, phys_addr, index); } void PrefetchSingleLine::flush() { - // Since the address is an unsigned int, the next index will be negative and will force the prefetcher - // to refill + // Mark the buffer empty. buffer_start_addr=-1 is kept for legacy callers, but + // it is NOT sufficient on its own: the fetch() fast-path index is unsigned and + // wraps to addr+1, so a fetch in the first bytes of the address space + // (addr <= 0x0b with a 16-byte line) would still hit in-window and read the + // stale line (stale opcode -> bogus next-PC). buffer_valid=false forces a + // real refill on the next fetch. Also drop any in-flight prefetch insn so a + // stale resume cannot fire. this->buffer_start_addr = -1; + this->buffer_valid = false; + this->prefetch_insn = NULL; } void PrefetchSingleLine::handle_stall(void (*callback)(PrefetchSingleLine *), iss_insn_t *current_insn)