From 201fada52ea0c259a6c0d97f1b1d67d0c0a0f111 Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 12 Jun 2026 11:57:15 +0200 Subject: [PATCH 01/28] feat(cv32e40p): add standalone GVSOC platform and core config Add the CV32E40P standalone GVSOC target for UVM co-simulation via the RVVI bridge, plus the CV32E40P core model in pulp_cores.py. The platform exposes the CV32E40P RTL generics (corev_pulp, fpu, zfinx, corev_cluster, core_version, num_mhpmcounters) as gvrun parameters and wires the core to a router with the CV32E40P memory map (RAM, virtual stdout/timer sinks, debug ROM, virtual exit). The core model computes misa and mimpid from the configured generics to match cv32e40p_cs_registers.sv, and writes the vendor CSR reset values and write masks consumed by Cv32e40pCsr::build() in csr_cv32e40p.cpp. --- cv32e40p-standalone.py | 132 +++++++++++++++++++++++++++++++++++++ pulp/cpu/iss/pulp_cores.py | 123 ++++++++++++++++++++++++++++++++++ 2 files changed, 255 insertions(+) create mode 100644 cv32e40p-standalone.py diff --git a/cv32e40p-standalone.py b/cv32e40p-standalone.py new file mode 100644 index 00000000..eeeb65b9 --- /dev/null +++ b/cv32e40p-standalone.py @@ -0,0 +1,132 @@ +# +# Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and +# University of Bologna +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P standalone GVSOC target for UVM co-simulation with RVVI bridge. +# +# Follows the pattern of tutorial 17 (how_to_control_gvsoc_from_an_external_simulator) +# adapted for CV32E40P (ri5cy/PULP FC core) with correct memory map. +# +# Memory map (from cv32e40p/bsp/): +# 0x00000000 4MB Main RAM (entry point 0x00000080) +# 0x10000000 256B Virtual STDOUT (write-only sink) +# 0x15000000 256B Virtual TIMER (write-only sink) +# 0x1A110800 16KB Debug ROM (read-only sink) +# 0x20000000 256B Virtual EXIT (write-only sink) + +import memory.memory +import vp.clock_domain +import interco.router +import utils.loader.loader +import gvsoc.systree +import gvsoc.runner +from gvrun.parameter import TargetParameter +from pulp.cpu.iss.pulp_cores import cv32e40p +from pulp.cv32e40p_exit.cv32e40p_exit_device import Cv32e40pExitDevice + + +class Cv32e40pSoc(gvsoc.systree.Component): + + def __init__(self, parent, name, binary, corev_pulp, fpu, zfinx, corev_cluster, core_version, + num_mhpmcounters=1): + super().__init__(parent, name) + + # Main interconnect + ico = interco.router.Router(self, 'ico') + + # 4MB main RAM @ 0x00000000 (entry point 0x00000080) + mem = memory.memory.Memory(self, 'mem', size=0x00400000, init=False) + ico.o_MAP(mem.i_INPUT(), 'mem', base=0x00000000, size=0x00400000, rm_base=True) + + # Virtual STDOUT sink @ 0x10000000 (256B) + stdout_mem = memory.memory.Memory(self, 'stdout', size=0x100, init=False) + ico.o_MAP(stdout_mem.i_INPUT(), 'stdout', base=0x10000000, size=0x100, rm_base=True) + + # Virtual TIMER sink @ 0x15000000 (256B) + timer_mem = memory.memory.Memory(self, 'timer', size=0x100, init=False) + ico.o_MAP(timer_mem.i_INPUT(), 'timer', base=0x15000000, size=0x100, rm_base=True) + + # Debug ROM sink @ 0x1A110800 (16KB) + debug_mem = memory.memory.Memory(self, 'debug_rom', size=0x4000, init=False) + ico.o_MAP(debug_mem.i_INPUT(), 'debug_rom', base=0x1A110800, size=0x4000, rm_base=True) + + # Virtual EXIT device @ 0x20000000 (256B) + # Terminates GVSOC when the program writes exit_valid to offset +0x04 + exit_dev = Cv32e40pExitDevice(self, 'exit') + ico.o_MAP(exit_dev.i_INPUT(), 'exit', base=0x20000000, size=0x100, rm_base=True) + + # CV32E40P core: uses the cv32e40p model which sets CONFIG_ISS_CORE=cv32e40p + # and computes misa/mimpid from fpu/zfinx/pulpv2 to match the RTL configuration. + core = cv32e40p(self, 'core', fetch_enable=False, boot_addr=0x00000080, + cluster_id=0, pulpv2=corev_pulp, fpu=fpu, zfinx=zfinx, + corev_cluster=corev_cluster, core_version=core_version, + num_mhpmcounters=num_mhpmcounters) + core.o_FETCH(ico.i_INPUT()) + core.o_DATA(ico.i_INPUT()) + + + # ELF loader: loads binary into RAM, signals core entry point and fetch enable + loader = utils.loader.loader.ElfLoader(self, 'loader', binary=binary) + loader.o_OUT(ico.i_INPUT()) + loader.o_START(core.i_FETCHEN()) + loader.o_ENTRY(core.i_ENTRY()) + + +# Wrapping component that attaches a clock generator, following tutorial 17 pattern. +class Cv32e40p(gvsoc.systree.Component): + + def __init__(self, parent, name=None): + super().__init__(parent, name) + + binary = TargetParameter( + self, name='binary', value=None, description='ELF binary to simulate' + ).get_value() + + # Core configuration parameters — match CV32E40P RTL generics. + # Pass via --parameter on the gvrun command line (default = base config). + corev_pulp = TargetParameter(self, name='corev_pulp', value=False, + description='Enable PULP extensions (COREV_PULP RTL param)').get_value() + fpu = TargetParameter(self, name='fpu', value=False, + description='Enable FPU (FPU RTL param)').get_value() + zfinx = TargetParameter(self, name='zfinx', value=False, + description='FPU uses integer regfile (ZFINX RTL param)').get_value() + corev_cluster = TargetParameter(self, name='corev_cluster', value=False, + description='Cluster variant (COREV_CLUSTER RTL param)').get_value() + core_version = TargetParameter(self, name='core_version', value=2, + description='ISA version: 1 for legacy PULP, 2 for CORE-V v2').get_value() + num_mhpmcounters = TargetParameter(self, name='num_mhpmcounters', value=1, + description='Number of HPM counters (NUM_MHPMCOUNTERS RTL param)').get_value() + + clock = vp.clock_domain.Clock_domain(self, 'clock', frequency=50000000) + soc = Cv32e40pSoc(self, 'soc', binary, + corev_pulp=corev_pulp, fpu=fpu, + zfinx=zfinx, corev_cluster=corev_cluster, + core_version=core_version, + num_mhpmcounters=num_mhpmcounters) + clock.o_CLOCK(soc.i_CLOCK()) + + +# Top target that gvrun will instantiate (mirrors tutorial 17 Target class structure) +class Target(gvsoc.runner.Target): + + description = "CV32E40P standalone for UVM co-simulation" + model = Cv32e40p + name = "cv32e40p-standalone" diff --git a/pulp/cpu/iss/pulp_cores.py b/pulp/cpu/iss/pulp_cores.py index e83549b8..b3447d05 100644 --- a/pulp/cpu/iss/pulp_cores.py +++ b/pulp/cpu/iss/pulp_cores.py @@ -16,10 +16,15 @@ # limitations under the License. # +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + import cpu.iss.riscv from cpu.iss.isa_gen.isa_riscv_gen import * from cpu.iss.isa_gen.isa_smallfloats import * from cpu.iss.isa_gen.isa_pulpv2 import * +from cpu.iss.isa_gen.isa_cv32e40pv2 import * from cpu.iss.isa_gen.isa_pulpnn import PulpNn @@ -109,3 +114,121 @@ def __init__(self, parent, name, fetch_enable: bool=False, boot_addr: int=0, clu super().__init__(parent, name, isa=_fc_isa, cluster_id=cluster_id, core_id=0, fetch_enable=fetch_enable, boot_addr=boot_addr) + + +def _build_cv32e40p_isa(name, pulpv2=True, fpu=True, zfinx=False, core_version=2): + """Build the CV32E40P ISA. + + fpu=True → rv32imfc (F instructions in decoder, regardless of zfinx). + fpu=False → rv32imc (no F instructions). + + ZFINX uses the same F-extension opcodes but routes them to GPR via + ISS_SINGLE_REGFILE (set in RiscvCommon.__init__). The MISA F-bit and + mstatus FS mask are controlled separately by fpu_in_isa in add_properties. + """ + + # ZFINX needs F instructions in the decoder (routed to GPR by ISS_SINGLE_REGFILE). + base_isa = 'rv32imfc' if fpu else 'rv32imc' + + extensions = [] + if pulpv2: + if core_version == 2: + extensions.append(CoreV2()) + else: + extensions.append(PulpV2()) + + isa = cpu.iss.isa_gen.isa_riscv_gen.RiscvIsa(name, base_isa, extensions=extensions) + + return isa + + +class cv32e40p(cpu.iss.riscv.RiscvCommon): + + def __init__(self, parent, name, fetch_enable: bool=False, boot_addr: int=0, cluster_id: int=31, core_id: int=0, + pulpv2: bool=True, fpu: bool=True, zfinx: bool=False, corev_cluster: bool=False, core_version: int=2, + num_mhpmcounters: int=1): + """ + CV32E40P core model. + + Parameters + ---------- + pulpv2 : bool + Enable PULP extensions (COREV_PULP). Default True. + fpu : bool + Enable FPU (rv32imfc ISA, misa bit F set). Default True. + zfinx : bool + FPU uses integer registers (Zfinx). Clears misa bit F. Default False. + corev_cluster : bool + Cluster variant (COREV_CLUSTER). Contributes to mimpid. Default False. + core_version : int + 1 for legacy PULP v1 encodings, 2 for CORE-V v2 (custom-0/1/2). Default 2. + num_mhpmcounters : int + Number of HPM counters (RTL NUM_MHPMCOUNTERS param). Default 1. + """ + # Unique ISA name per instance/configuration to avoid generator collisions + isa_name = f'cv32e40p_{name}_v{core_version}_{"f" if fpu else "nof"}_{"z" if zfinx else "noz"}' + isa = _build_cv32e40p_isa(isa_name, pulpv2=pulpv2, fpu=fpu, zfinx=zfinx, core_version=core_version) + + # misa: MXL=1(RV32) | I(bit8) | M(bit12) | C(bit2) [| F(bit5) if fpu and not zfinx] [| X(bit23) if pulpv2] + _MISA_BASE = 0x40001104 # RV32 | I | M | C + fpu_in_isa = fpu and not zfinx + misa = _MISA_BASE | (0x20 if fpu_in_isa else 0) | (0x00800000 if pulpv2 else 0) + + # mimpid = (FPU || COREV_PULP || COREV_CLUSTER) ? 1 : 0 (RTL cv32e40p_cs_registers.sv) + mimpid = 0x1 if (fpu or pulpv2 or corev_cluster) else 0x0 + + super().__init__(parent, name, isa=isa, + riscv_dbg_unit=True, fetch_enable=fetch_enable, boot_addr=boot_addr, + first_external_pcer=12, debug_handler=0x1a110800, misa=misa, core="riscv", + cluster_id=cluster_id, core_id=core_id, wrapper="pulp/cpu/iss/default_iss_wrapper.cpp", + scoreboard=True, timed=True, handle_misaligned=True, zfinx=zfinx, + riscv_exceptions=True) + + # CV32E40P / PULP vendor CSR values — written to JSON config. + # These are read by Cv32e40pCsr::build() in csr_cv32e40p.cpp. + # Values derived from cv32e40p_cs_registers.sv and cv32e40p_pkg.sv. + # mstatus effective write mask — matches RTL always_ff forcing (PULP_SECURE=0). + # RTL cv32e40p_cs_registers.sv:1222-1230 forces MPP=M, MPRV=0, UIE=0, UPIE=0. + # Only MIE(3) + MPIE(7) are writable. With FPU: add FS(14:13). + mstatus_mask = 0x6088 if fpu_in_isa else 0x0088 + # mcountinhibit: CY(0) + IR(2) + HPM3..HPM(2+N) — bit 1 always reserved + # With N=1: mask=0x0D (bits 0,2,3). With N=29: mask=0xFFFFFFFD (all except bit 1). + mcountinhibit_mask = 0x5 | (((1 << num_mhpmcounters) - 1) << 3) + self.add_properties({ + 'mvendorid_value': 0x602, + 'marchid_value': 0x4, + 'mimpid': mimpid, + 'fpu_in_isa': fpu_in_isa, + 'mtvec_reset': 0x1, + 'mtvec_write_mask': 0xFFFFFF01, # bits[7:1] hardwired 0 + 'mcause_mask': 0x8000001F, # bit[31] + bits[4:0] + 'mcountinhibit_mask': mcountinhibit_mask, + 'mstatus_write_mask': mstatus_mask, # MPP/MPIE/MIE [+FS if FPU] + 'mie_write_mask': 0xFFFF0888, # IRQ_MASK + 'mtval_write_mask': 0x00000000, # CV32E40P mtval is hardwired to 0 + 'tdata1_reset': 0x28001040, # type=2,dmode=1,action=1,m=1,u=0 (no U-mode) + 'tdata1_write_mask': 0x00000000, # writable ONLY from Debug Mode (RTL: tmatch_control_we = csr_we_int & debug_mode_i) + 'tdata2_write_mask': 0x00000000, # writable ONLY from Debug Mode (RTL: tmatch_value_we = csr_we_int & debug_mode_i) + 'tinfo_reset': 0x4, # bit[2] = mcontrol supported + 'num_mhpmcounters': num_mhpmcounters, + 'num_hpm_events': 16, + 'pulpv2': pulpv2, # COREV_PULP — gates UHARTID/PRIVLV + 'zfinx': zfinx, # ZFINX mode — gates CSR_ZFINX (0xCD2) + }) + + self.add_c_flags([ + "-DPIPELINE_STALL_THRESHOLD=1", + "-DCONFIG_ISS_CORE=cv32e40p", + f"-DCONFIG_GVSOC_CORE_VERSION={core_version}", + "-DCONFIG_GVSOC_ISS_HWLOOP=1", + "-DCONFIG_GVSOC_ISS_CV32E40P=1", + ]) + + # CV32E40P-specific CSR subclass (Cv32e40pCsr) — must be compiled + # into the ISS model .so. The base riscv.py add_sources() list + # doesn't include it because it's CV32E40P-specific. + self.add_sources([ + "cpu/iss/src/cv32e40p/csr_cv32e40p.cpp", + "cpu/iss/src/cv32e40p/irq_cv32e40p.cpp", + "cpu/iss/src/cv32e40p/core_cv32e40p.cpp", + ]) From c61df88cce82aff35a31b793b0022eeb773566ca Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 12 Jun 2026 11:57:26 +0200 Subject: [PATCH 02/28] feat(cv32e40p_exit): add virtual exit device Add the CV32E40P virtual exit device that terminates the GVSOC simulation when the test program writes exit_valid at 0x2000_0004, mirroring the cv32e40p virtual peripheral status flags described in test_programs.rst. Includes the C++ device model, its Python wrapper, and the CMake wiring under pulp/cv32e40p_exit/. --- pulp/CMakeLists.txt | 1 + pulp/cv32e40p_exit/CMakeLists.txt | 3 + pulp/cv32e40p_exit/cv32e40p_exit_device.cpp | 146 ++++++++++++++++++++ pulp/cv32e40p_exit/cv32e40p_exit_device.py | 46 ++++++ 4 files changed, 196 insertions(+) create mode 100644 pulp/cv32e40p_exit/CMakeLists.txt create mode 100644 pulp/cv32e40p_exit/cv32e40p_exit_device.cpp create mode 100644 pulp/cv32e40p_exit/cv32e40p_exit_device.py diff --git a/pulp/CMakeLists.txt b/pulp/CMakeLists.txt index 80bb94de..cab57746 100644 --- a/pulp/CMakeLists.txt +++ b/pulp/CMakeLists.txt @@ -19,3 +19,4 @@ add_subdirectory(datamover) add_subdirectory(snitch) add_subdirectory(redmule) add_subdirectory(pcie_vfio_bridge) +add_subdirectory(cv32e40p_exit) diff --git a/pulp/cv32e40p_exit/CMakeLists.txt b/pulp/cv32e40p_exit/CMakeLists.txt new file mode 100644 index 00000000..3de331b8 --- /dev/null +++ b/pulp/cv32e40p_exit/CMakeLists.txt @@ -0,0 +1,3 @@ +vp_model(NAME pulp.cv32e40p_exit.cv32e40p_exit_device + SOURCES "cv32e40p_exit_device.cpp" + ) diff --git a/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp b/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp new file mode 100644 index 00000000..b3296d2f --- /dev/null +++ b/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp @@ -0,0 +1,146 @@ +/* + * Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and + * University of Bologna + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +/* + * CV32E40P virtual exit device for GVSOC. + * + * Implements the cv32e40p virtual peripheral status flags (test_programs.rst): + * + * Offset 0x00 (0x2000_0000): test_passed / test_failed flags (write-only sink) + * Offset 0x04 (0x2000_0004): assert exit_valid → terminates GVSOC simulation + * exit_value = wdata + * Offset 0x08 (0x2000_0008): signature_start_address (write-only sink) + * Offset 0x0C (0x2000_000C): signature_end_address (write-only sink) + * Offset 0x10 (0x2000_0010): signature write trigger (write-only sink) + * also asserts exit_valid with exit_value = 0 + */ + +#include + +#include +#include + +#define VP_STATUS_FLAGS_OFFSET 0x00 +#define VP_EXIT_VALID_OFFSET 0x04 +#define VP_SIG_START_OFFSET 0x08 +#define VP_SIG_END_OFFSET 0x0C +#define VP_SIG_WRITE_OFFSET 0x10 + +class Cv32e40pExitDevice : public vp::Component +{ +public: + Cv32e40pExitDevice(vp::ComponentConf &config); + +private: + static vp::IoReqStatus handle_req(vp::Block *__this, vp::IoReq *req); + + vp::IoSlave input_itf; + vp::Trace trace; +}; + +Cv32e40pExitDevice::Cv32e40pExitDevice(vp::ComponentConf &config) + : vp::Component(config) +{ + this->input_itf.set_req_meth(&Cv32e40pExitDevice::handle_req); + this->new_slave_port("input", &this->input_itf); + this->traces.new_trace("trace", &this->trace, vp::DEBUG); +} + +vp::IoReqStatus Cv32e40pExitDevice::handle_req(vp::Block *__this, vp::IoReq *req) +{ + Cv32e40pExitDevice *_this = (Cv32e40pExitDevice *)__this; + + if (!req->get_is_write()) + { + /* All registers are write-only — return 0 on read */ + memset(req->get_data(), 0, req->get_size()); + return vp::IO_REQ_OK; + } + + uint32_t offset = (uint32_t)req->get_addr(); + uint32_t wdata = (req->get_size() == 4) ? *(uint32_t *)req->get_data() : 0; + + switch (offset) + { + case VP_STATUS_FLAGS_OFFSET: + if (wdata == 123456789U) /* 0x075BCD15 — TEST PASSED (same magic as UVM VP) */ + { + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "vp_status write: wdata=0x%08x — TEST PASSED, stopping simulation\n", wdata); + fprintf(stdout, "[cv32e40p_exit] tests_passed=1 exit_value=0x00000000\n"); + fflush(stdout); + _this->time.get_engine()->quit(0); + } + else if (wdata == 1U) /* TEST FAILED */ + { + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "vp_status write: wdata=0x%08x — TEST FAILED, stopping simulation\n", wdata); + fprintf(stdout, "[cv32e40p_exit] tests_failed=1 exit_value=0x00000001\n"); + fflush(stdout); + _this->time.get_engine()->quit(1); + } + else + { + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "vp_status write: wdata=0x%08x (unrecognized status flag — ignored)\n", wdata); + } + break; + + case VP_EXIT_VALID_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "exit_valid asserted: exit_value=0x%08x — stopping simulation\n", wdata); + fprintf(stdout, "[cv32e40p_exit] exit_valid=1 exit_value=0x%08x\n", wdata); + fflush(stdout); + _this->time.get_engine()->quit((int32_t)wdata); + break; + + case VP_SIG_START_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "signature_start_address=0x%08x (not implemented)\n", wdata); + break; + + case VP_SIG_END_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "signature_end_address=0x%08x (not implemented)\n", wdata); + break; + + case VP_SIG_WRITE_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "signature write triggered — stopping simulation (exit_value=0)\n"); + fprintf(stdout, "[cv32e40p_exit] signature write → exit_valid=1 exit_value=0x00000000\n"); + fflush(stdout); + _this->time.get_engine()->quit(0); + break; + + default: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "unknown offset 0x%02x wdata=0x%08x — ignored\n", offset, wdata); + break; + } + + return vp::IO_REQ_OK; +} + +extern "C" vp::Component *gv_new(vp::ComponentConf &config) +{ + return new Cv32e40pExitDevice(config); +} diff --git a/pulp/cv32e40p_exit/cv32e40p_exit_device.py b/pulp/cv32e40p_exit/cv32e40p_exit_device.py new file mode 100644 index 00000000..cc98fdfa --- /dev/null +++ b/pulp/cv32e40p_exit/cv32e40p_exit_device.py @@ -0,0 +1,46 @@ +# +# Copyright (C) 2020 GreenWaves Technologies, SAS, ETH Zurich and +# University of Bologna +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# Python wrapper for the CV32E40P virtual exit device. +# Terminates GVSOC simulation when the test program writes to 0x2000_0004. + +import gvsoc.systree as st + + +class Cv32e40pExitDevice(st.Component): + """ + CV32E40P virtual peripheral status flags device. + + Memory map (base-relative): + +0x00 VP status flags (test_passed / test_failed — sink) + +0x04 exit_valid write → quits GVSOC simulation with exit_value = wdata + +0x08 sig_start_addr (sink) + +0x0C sig_end_addr (sink) + +0x10 sig_write trigger → quits GVSOC simulation with exit_value = 0 + """ + + def __init__(self, parent, name): + super().__init__(parent, name) + self.set_component('pulp.cv32e40p_exit.cv32e40p_exit_device') + + def i_INPUT(self) -> st.SlaveItf: + return st.SlaveItf(self, 'input', signature='io') From e0f1e3695f46bc5e837c077aa983d1b1d7575de3 Mon Sep 17 00:00:00 2001 From: mpaci Date: Wed, 15 Jul 2026 16:34:48 +0200 Subject: [PATCH 03/28] feat(cv32e40p): serve unmapped addresses from a background sparse memory Map a new byte-granular sparse-memory component as the interconnect default route (size=0 mapping, absolute addresses). The UVM testbench serves the whole address space from a zero-default sparse memory, so out-of-map accesses must behave the same on the reference platform: reads return 0 until written, writes persist. Without this, stray stores were silently dropped (later readback diverged) and stray fetches faulted with mcause=1 where the RTL fetches 0 and traps illegal instruction (mcause=2). Also: - exit device: retain writes in a 256B backing store so write-then-read (e.g. of the signature addresses) returns the stored value, as it does through the testbench memory. - debug ROM: shrink 16KB -> 4KB to match the linker script dbg region; accesses past it now fall through to the background memory. --- cv32e40p-standalone.py | 23 +++- pulp/CMakeLists.txt | 1 + pulp/cv32e40p_exit/cv32e40p_exit_device.cpp | 29 +++-- pulp/cv32e40p_sparse_mem/CMakeLists.txt | 3 + .../cv32e40p_sparse_mem.cpp | 104 ++++++++++++++++++ .../cv32e40p_sparse_mem.py | 36 ++++++ 6 files changed, 184 insertions(+), 12 deletions(-) create mode 100644 pulp/cv32e40p_sparse_mem/CMakeLists.txt create mode 100644 pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.cpp create mode 100644 pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.py diff --git a/cv32e40p-standalone.py b/cv32e40p-standalone.py index eeeb65b9..1af7f939 100644 --- a/cv32e40p-standalone.py +++ b/cv32e40p-standalone.py @@ -29,8 +29,10 @@ # 0x00000000 4MB Main RAM (entry point 0x00000080) # 0x10000000 256B Virtual STDOUT (write-only sink) # 0x15000000 256B Virtual TIMER (write-only sink) -# 0x1A110800 16KB Debug ROM (read-only sink) -# 0x20000000 256B Virtual EXIT (write-only sink) +# 0x1A110800 4KB Debug ROM (linker script `dbg` region) +# 0x20000000 256B Virtual EXIT (terminates the simulation) +# everywhere else background sparse memory (default route: reads 0 until +# written, writes persist - same as the UVM testbench) import memory.memory import vp.clock_domain @@ -41,6 +43,7 @@ from gvrun.parameter import TargetParameter from pulp.cpu.iss.pulp_cores import cv32e40p from pulp.cv32e40p_exit.cv32e40p_exit_device import Cv32e40pExitDevice +from pulp.cv32e40p_sparse_mem.cv32e40p_sparse_mem import Cv32e40pSparseMem class Cv32e40pSoc(gvsoc.systree.Component): @@ -64,15 +67,25 @@ def __init__(self, parent, name, binary, corev_pulp, fpu, zfinx, corev_cluster, timer_mem = memory.memory.Memory(self, 'timer', size=0x100, init=False) ico.o_MAP(timer_mem.i_INPUT(), 'timer', base=0x15000000, size=0x100, rm_base=True) - # Debug ROM sink @ 0x1A110800 (16KB) - debug_mem = memory.memory.Memory(self, 'debug_rom', size=0x4000, init=False) - ico.o_MAP(debug_mem.i_INPUT(), 'debug_rom', base=0x1A110800, size=0x4000, rm_base=True) + # Debug ROM @ 0x1A110800 (4KB, matches the linker script `dbg` region) + debug_mem = memory.memory.Memory(self, 'debug_rom', size=0x1000, init=False) + ico.o_MAP(debug_mem.i_INPUT(), 'debug_rom', base=0x1A110800, size=0x1000, rm_base=True) # Virtual EXIT device @ 0x20000000 (256B) # Terminates GVSOC when the program writes exit_valid to offset +0x04 exit_dev = Cv32e40pExitDevice(self, 'exit') ico.o_MAP(exit_dev.i_INPUT(), 'exit', base=0x20000000, size=0x100, rm_base=True) + # Background sparse memory: default route for everything not mapped + # above (size=0 mapping). The UVM testbench serves the whole address + # space from a zero-default sparse memory; without this, out-of-map + # stores are dropped (readback diverges) and out-of-map fetches fault + # with mcause=1 where the RTL executes 0 and traps illegal (mcause=2). + # Absolute addresses are forwarded (rm_base=False) so the store is + # indexed like the testbench's. + bg_mem = Cv32e40pSparseMem(self, 'background_mem') + ico.o_MAP(bg_mem.i_INPUT(), 'background', base=0x00000000, size=0, rm_base=False) + # CV32E40P core: uses the cv32e40p model which sets CONFIG_ISS_CORE=cv32e40p # and computes misa/mimpid from fpu/zfinx/pulpv2 to match the RTL configuration. core = cv32e40p(self, 'core', fetch_enable=False, boot_addr=0x00000080, diff --git a/pulp/CMakeLists.txt b/pulp/CMakeLists.txt index cab57746..54ba73ed 100644 --- a/pulp/CMakeLists.txt +++ b/pulp/CMakeLists.txt @@ -20,3 +20,4 @@ add_subdirectory(snitch) add_subdirectory(redmule) add_subdirectory(pcie_vfio_bridge) add_subdirectory(cv32e40p_exit) +add_subdirectory(cv32e40p_sparse_mem) diff --git a/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp b/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp index b3296d2f..3a9330f5 100644 --- a/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp +++ b/pulp/cv32e40p_exit/cv32e40p_exit_device.cpp @@ -25,13 +25,18 @@ * * Implements the cv32e40p virtual peripheral status flags (test_programs.rst): * - * Offset 0x00 (0x2000_0000): test_passed / test_failed flags (write-only sink) + * Offset 0x00 (0x2000_0000): test_passed / test_failed flags * Offset 0x04 (0x2000_0004): assert exit_valid → terminates GVSOC simulation * exit_value = wdata - * Offset 0x08 (0x2000_0008): signature_start_address (write-only sink) - * Offset 0x0C (0x2000_000C): signature_end_address (write-only sink) - * Offset 0x10 (0x2000_0010): signature write trigger (write-only sink) + * Offset 0x08 (0x2000_0008): signature_start_address + * Offset 0x0C (0x2000_000C): signature_end_address + * Offset 0x10 (0x2000_0010): signature write trigger * also asserts exit_valid with exit_value = 0 + * + * Every write is also retained and readable back: in the UVM testbench this + * region is ordinary sparse memory that the virtual peripheral snoops, so a + * write-then-read (e.g. of the signature addresses) returns the stored value + * there and must do the same here. */ #include @@ -55,6 +60,10 @@ class Cv32e40pExitDevice : public vp::Component vp::IoSlave input_itf; vp::Trace trace; + + /* Backing store (256B region): writes persist and read back, like the + * testbench memory under the virtual peripheral. */ + uint8_t mem[0x100] = {}; }; Cv32e40pExitDevice::Cv32e40pExitDevice(vp::ComponentConf &config) @@ -69,14 +78,20 @@ vp::IoReqStatus Cv32e40pExitDevice::handle_req(vp::Block *__this, vp::IoReq *req { Cv32e40pExitDevice *_this = (Cv32e40pExitDevice *)__this; + uint32_t offset = (uint32_t)req->get_addr(); + if (!req->get_is_write()) { - /* All registers are write-only — return 0 on read */ - memset(req->get_data(), 0, req->get_size()); + if (offset + req->get_size() <= sizeof(_this->mem)) + memcpy(req->get_data(), &_this->mem[offset], req->get_size()); + else + memset(req->get_data(), 0, req->get_size()); return vp::IO_REQ_OK; } - uint32_t offset = (uint32_t)req->get_addr(); + if (offset + req->get_size() <= sizeof(_this->mem)) + memcpy(&_this->mem[offset], req->get_data(), req->get_size()); + uint32_t wdata = (req->get_size() == 4) ? *(uint32_t *)req->get_data() : 0; switch (offset) diff --git a/pulp/cv32e40p_sparse_mem/CMakeLists.txt b/pulp/cv32e40p_sparse_mem/CMakeLists.txt new file mode 100644 index 00000000..d5c8e2ce --- /dev/null +++ b/pulp/cv32e40p_sparse_mem/CMakeLists.txt @@ -0,0 +1,3 @@ +vp_model(NAME pulp.cv32e40p_sparse_mem.cv32e40p_sparse_mem + SOURCES "cv32e40p_sparse_mem.cpp" + ) diff --git a/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.cpp b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.cpp new file mode 100644 index 00000000..88f9dac9 --- /dev/null +++ b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.cpp @@ -0,0 +1,104 @@ +/* + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +/* + * Background sparse memory for the CV32E40P standalone platform. + * + * Mapped as the interconnect's default route (size=0 mapping), it serves + * every access that no explicit device claims. The UVM testbench answers the + * whole address space from a sparse memory model - a never-written location + * reads 0, a write persists - so the reference platform must do the same or + * stray software accesses diverge from the RTL: + * + * - an out-of-map store followed by a load must return the stored value, + * not 0 (the store used to be dropped); + * - a fetch from an unmapped address must return 0x00000000, which decodes + * as an illegal instruction (mcause=2) exactly like the RTL executing + * the testbench's zero response - not an instruction access fault + * (mcause=1) raised before execution. + * + * Contract with the testbench: this component reads 0 for never-written + * bytes, which matches the cv32e40p environment only because its memory + * model defaults to zero-fill. A testbench configured to random-fill would + * diverge from any zero-defaulting reference by construction. + */ + +#include +#include + +#include +#include + +class Cv32e40pSparseMem : public vp::Component +{ +public: + Cv32e40pSparseMem(vp::ComponentConf &config); + +private: + static vp::IoReqStatus handle_req(vp::Block *__this, vp::IoReq *req); + + vp::IoSlave input_itf; + vp::Trace trace; + + /* Byte-granular backing store: only written bytes are kept. */ + std::unordered_map store; +}; + +Cv32e40pSparseMem::Cv32e40pSparseMem(vp::ComponentConf &config) + : vp::Component(config) +{ + this->input_itf.set_req_meth(&Cv32e40pSparseMem::handle_req); + this->new_slave_port("input", &this->input_itf); + this->traces.new_trace("trace", &this->trace, vp::DEBUG); +} + +vp::IoReqStatus Cv32e40pSparseMem::handle_req(vp::Block *__this, vp::IoReq *req) +{ + Cv32e40pSparseMem *_this = (Cv32e40pSparseMem *)__this; + + uint64_t addr = req->get_addr(); + uint64_t size = req->get_size(); + uint8_t *data = req->get_data(); + + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "background access (addr: 0x%llx, size: 0x%llx, is_write: %d)\n", + addr, size, req->get_is_write()); + + if (req->get_is_write()) + { + for (uint64_t i = 0; i < size; i++) + _this->store[addr + i] = data[i]; + } + else + { + for (uint64_t i = 0; i < size; i++) + { + auto it = _this->store.find(addr + i); + data[i] = (it != _this->store.end()) ? it->second : 0; + } + } + + return vp::IO_REQ_OK; +} + +extern "C" vp::Component *gv_new(vp::ComponentConf &config) +{ + return new Cv32e40pSparseMem(config); +} diff --git a/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.py b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.py new file mode 100644 index 00000000..5eed6ede --- /dev/null +++ b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem.py @@ -0,0 +1,36 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# Python wrapper for the CV32E40P background sparse memory. +# Mapped as the interconnect's default route: serves every access no explicit +# device claims (never-written bytes read 0, writes persist), matching the UVM +# testbench sparse memory model. + +import gvsoc.systree as st + + +class Cv32e40pSparseMem(st.Component): + + def __init__(self, parent, name): + super().__init__(parent, name) + self.set_component('pulp.cv32e40p_sparse_mem.cv32e40p_sparse_mem') + + def i_INPUT(self) -> st.SlaveItf: + return st.SlaveItf(self, 'input', signature='io') From f009204f0264055c137820296376dd677f49f6fb Mon Sep 17 00:00:00 2001 From: mpaci Date: Thu, 16 Jul 2026 18:13:49 +0200 Subject: [PATCH 04/28] feat(cv32e40p): add iss_v2 bring-up recipe and spike target MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit First step of the CV32E40P migration to the modular iss_v2 core: - pulp/cpu/iss/cv32e40p_v2.py: core recipe on RiscvCommon with the generic v2 slots (IrqExternal, Event, Csr, ExecInOrder with scoreboard and in-order commit, LsuV2, scoreboarded Regfile, Hwloop) plus the CoreV ISA subset. The CV32E40P-specific CSR map, counters and trap behaviour will come in as slot overrides on top of this base, same layering as Ri5ky. - cv32e40p-v2-spike.py: minimal bring-up platform. LsuV2 drives the fetch/data ports with the io_v2 protocol, so the platform lives on the io_v2 plane (router_v2, memory_v3, loader_v2) following the Ri5ky testbench layout, and reuses its MMIO peripheral (putchar, exit) at 0x10000000. Integer bring-up ISA is rv32im + CoreV: with C enabled the generated decode table still references the rvf compressed handlers even when they are inactive (no F), and nothing declares them without an FPU module — to be addressed separately. Validated with hand-encoded probes on the spike target: cv.lw post-increment (data + pointer) and a 5-iteration lp.setupi hardware loop, both passing. --- cv32e40p-v2-spike.py | 148 ++++++++++++++++++++++++++++++++++++ pulp/cpu/iss/cv32e40p_v2.py | 79 +++++++++++++++++++ 2 files changed, 227 insertions(+) create mode 100644 cv32e40p-v2-spike.py create mode 100644 pulp/cpu/iss/cv32e40p_v2.py diff --git a/cv32e40p-v2-spike.py b/cv32e40p-v2-spike.py new file mode 100644 index 00000000..03d4f6d1 --- /dev/null +++ b/cv32e40p-v2-spike.py @@ -0,0 +1,148 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 bring-up target. +# +# The iss_v2 LSU (LsuV2) drives the fetch/data ports with the io_v2 +# protocol, so the whole platform lives on the io_v2 plane: router_v2, +# memory_v3 and loader_v2, following the Ri5ky testbench layout +# (pulp/ri5ky/ri5ky_testbench.py). The MMIO peripheral is reused as-is +# from that testbench: putchar @ +0x0, exit @ +0x4. +# +# Integer-only for the bring-up: rv32imc + the CoreV (XPULP v2) subset. + +import vp.clock_domain +import gvsoc.systree +import gvsoc.runner +from gvrun.parameter import TargetParameter +from config_tree import Config, cfg_field +from memory.memory_v3 import Memory, MemoryV3Config +from interco.router_v2 import Router, RouterConfig, RouterMapping +from utils.loader.loader_v2 import ElfLoader +from pulp.ri5ky.ri5ky_mmio import Ri5kyMmio +from pulp.cpu.iss.cv32e40p_v2 import Cv32e40p as Cv32e40pV2Core, Cv32e40pConfig + + +class Cv32e40pSpikeConfig(Config): + """Configuration for the CV32E40P iss_v2 bring-up SoC. + + Minimal layout: + - mem at 0x0000_0000 (4 MB), entry point 0x80 as in the RTL testbench + - MMIO at 0x1000_0000 (4 KB): putchar @ +0, exit @ +4 + """ + + mem_base: int = cfg_field(default=0x0000_0000, fmt="hex", dump=True, desc=( + "Base address of the main memory" + )) + + mem_size: int = cfg_field(default=0x40_0000, fmt="hex", dump=True, desc=( + "Size of the main memory" + )) + + mmio_base: int = cfg_field(default=0x1000_0000, fmt="hex", dump=True, desc=( + "Base address of the MMIO peripheral" + )) + + mmio_size: int = cfg_field(default=0x1000, fmt="hex", dump=True, desc=( + "Size of the MMIO peripheral window" + )) + + boot_addr: int = cfg_field(default=0x80, fmt="hex", dump=True, desc=( + "Boot address (matches RTL BOOT_ADDR)" + )) + + core: Cv32e40pConfig = cfg_field(init=False, desc=( + "CV32E40P core configuration" + )) + + mem: MemoryV3Config = cfg_field(init=False, desc=( + "Backing memory configuration" + )) + + router: RouterConfig = cfg_field(init=False, desc=( + "Router configuration" + )) + + mem_mapping: RouterMapping = cfg_field(init=False, desc=( + "Address range of the main memory" + )) + + mmio_mapping: RouterMapping = cfg_field(init=False, desc=( + "Address range of the MMIO peripheral" + )) + + def __post_init__(self): + super().__post_init__() + # No compressed for now: with C enabled the generated decode table + # references the rvf c.flwsp/c.fswsp handlers even when they are + # inactive (no F), and without an FPU module nothing declares them. + self.core = Cv32e40pConfig(isa='rv32im', boot_addr=self.boot_addr) + self.mem = MemoryV3Config('mem', size=self.mem_size, atomics=False, latency=1) + self.router = RouterConfig(kind='bandwidth') + self.mem_mapping = RouterMapping(name='mem_mapping', + base=self.mem_base, size=self.mem_size) + self.mmio_mapping = RouterMapping(name='mmio_mapping', + base=self.mmio_base, size=self.mmio_size) + + +class Cv32e40pSpikeSoc(gvsoc.systree.Component): + + def __init__(self, parent, name, config: Cv32e40pSpikeConfig, binary): + super().__init__(parent, name, config=config) + + mem = Memory ( self, 'mem' , config=config.mem ) + mmio = Ri5kyMmio ( self, 'mmio' ) + ico = Router ( self, 'ico' , config=config.router ) + core = Cv32e40pV2Core( self, 'core' , config=config.core ) + loader = ElfLoader ( self, 'loader', binary=binary ) + + ico.o_MAP ( mem.i_INPUT() , mapping=config.mem_mapping ) + ico.o_MAP ( mmio.i_INPUT(), mapping=config.mmio_mapping ) + + # Three independent masters, one router input port each. + loader.o_OUT ( ico.i_INPUT(0) ) + loader.o_START ( core.i_FETCHEN() ) + loader.o_ENTRY ( core.i_ENTRY() ) + + core.o_FETCH ( ico.i_INPUT(1) ) + core.o_DATA ( ico.i_INPUT(2) ) + + +class Cv32e40p(gvsoc.systree.Component): + + def __init__(self, parent, name=None): + super().__init__(parent, name) + + binary = TargetParameter( + self, name='binary', value=None, description='ELF binary to simulate' + ).get_value() + + config = Cv32e40pSpikeConfig('soc') + + clock = vp.clock_domain.Clock_domain(self, 'clock', frequency=50000000) + soc = Cv32e40pSpikeSoc(self, 'soc', config, binary) + clock.o_CLOCK(soc.i_CLOCK()) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 bring-up" + model = Cv32e40p + name = "cv32e40p-v2-spike" diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py new file mode 100644 index 00000000..156bab16 --- /dev/null +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -0,0 +1,79 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +from __future__ import annotations + +from typing import Iterable +from gvsoc.systree import Component +from cpu.iss_v2.riscv import (RiscvCommon, IrqExternal, ExecInOrder, Regfile, + Csr, Event, LsuV2, Hwloop) +from cpu.iss.isa_gen.isa_gen import Isa, IsaSubset +from cpu.iss.isa_gen.isa_riscv_gen import RiscvIsa +from cpu.iss.isa_gen.isa_cv32e40pv2 import CoreV2 +from cpu.iss_v2.riscv_config import RiscvConfig + +isa_instances: dict[tuple[str, str], Isa] = {} + + +class Cv32e40pConfig(RiscvConfig): + pass + + +class Cv32e40p(RiscvCommon): + """CV32E40P on the iss_v2 modular core. + + Bring-up recipe: generic v2 slots plus the CoreV ISA subset. The + CV32E40P-specific CSR map, event counters and trap behaviour come in + as dedicated slot overrides on top of this base (same layering as + Ri5ky). + """ + + # Tag used in ISA-cache and generated ISA-class names (see Ri5ky). + isa_name: str = 'cv32e40p_v2' + + def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, + extra_extensions: Iterable[IsaSubset] = ()): + + cache_key = (type(self).isa_name, config.isa) + isa_instance: Isa | None = isa_instances.get(cache_key) + + if isa_instance is None: + extensions: list[IsaSubset] = [ + *extra_extensions, + CoreV2(), + ] + + isa_instance = RiscvIsa(f"{type(self).isa_name}_{config.isa}", + config.isa, extensions=extensions) + + isa_instances[cache_key] = isa_instance + + modules: dict[str, object] = { + 'irq': IrqExternal(), + 'event': Event(), + 'csr': Csr(), + 'exec': ExecInOrder(scoreboard=True, inorder_commit=True), + 'lsu': LsuV2(), + 'regfile': Regfile(scoreboard=True), + 'hwloop': Hwloop(), + } + + super().__init__(parent, name, config=config, isa=isa_instance, + modules=modules) From 9353d7a1878a5ef3cd826e1dd8c7fe6bf9d57e52 Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 17 Jul 2026 10:22:50 +0200 Subject: [PATCH 05/28] feat(cv32e40p): CSR personality on iss_v2 Cv32e40pCsr models the RTL CSR map on the v2 modular core (same layering as Ri5ky): M-mode-only strict CSR space (undeclared addresses raise illegal-instruction), RTL write masks and reset values, read-only machine information and PULP custom CSRs (uhartid/privlv/zfinx), hardware-loop CSRs readable but not writable through CSR instructions (the architectural LPEND is re-derived from the hwloop module), trigger CSRs with the RTL tie-offs, mcycle with inhibit-aware offset semantics, and fflags/frm/fcsr legality following the RTL fs_off signal across the FPU/ZFINX/no-FPU configurations. The recipe switches the irq slot to the RISC-V privileged scheme (the RTL implements mie/mip/mtvec with standard mcause codes, not the PULP event unit) and keys the ISA cache on the PULP extension set. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 137 ++++++++++ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 316 ++++++++++++++++++++++ pulp/cpu/iss/cv32e40p_v2.py | 73 ++++- 3 files changed, 517 insertions(+), 9 deletions(-) create mode 100644 cpu/iss_v2/include/cores/cv32e40p/csr.hpp create mode 100644 cpu/iss_v2/src/cores/cv32e40p/csr.cpp diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp new file mode 100644 index 00000000..0332cadc --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -0,0 +1,137 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include +#include + +/* Static personality configuration, set by the Python recipe + * (pulp/cpu/iss/cv32e40p_v2.py): + * CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA F extension present (0 for ZFINX) + * CONFIG_GVSOC_ISS_CV32E40P_ZFINX ZFINX variant + * CONFIG_GVSOC_ISS_CV32E40P_PULP COREV_PULP (XPULP) configuration + * CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS implemented HPM counters + */ +#ifndef CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA +#define CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA 0 +#endif +#ifndef CONFIG_GVSOC_ISS_CV32E40P_ZFINX +#define CONFIG_GVSOC_ISS_CV32E40P_ZFINX 0 +#endif +#ifndef CONFIG_GVSOC_ISS_CV32E40P_PULP +#define CONFIG_GVSOC_ISS_CV32E40P_PULP 1 +#endif +#ifndef CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS +#define CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS 1 +#endif + +/* CSR that is read-only from CSR instructions: any write attempt raises an + * illegal-instruction exception before the access happens, so the destination + * register is not written (matches the RTL decoder behaviour). */ +class Cv32e40pRoCsr : public CsrReg +{ +public: + bool check_access(Iss *iss, bool write, bool read) override; +}; + +/* fflags / frm / fcsr front-end. Access legality follows the RTL fs_off + * signal: rejected with an illegal-instruction exception while mstatus.FS + * is Off (FPU in the ISA), always rejected without an FPU, always granted + * for ZFINX — see fp_access_illegal(). The register content lives in the + * base class fcsr field; the value mapping is done by the registered + * callback. */ +class Cv32e40pFpCsr : public CsrAbtractReg +{ +public: + bool check_access(Iss *iss, bool write, bool read) override; +}; + +/* Hardware-loop CSR front-end (lpstart/lpend/lpcount). Readable via CSR + * instructions, but the RTL decoder raises illegal-instruction on any CSR + * write to them (they are only programmed through the cv.* instructions). */ +class Cv32e40pHwloopCsr : public CsrAbtractReg +{ +public: + bool check_access(Iss *iss, bool write, bool read) override; +}; + +class Cv32e40pCsr : public Csr +{ +public: + Cv32e40pCsr(Iss &iss); + + void start(); + void reset(bool active); + + /* FP CSR access legality: illegal while mstatus.FS == Off (00). */ + inline bool fp_access_illegal(); + + /* Promote mstatus.FS to Dirty (11) on FP state change. The RTL forces it + * on FP regfile writes, fflags updates and FP-CSR writes when the FPU is + * in the ISA (FPU=1, ZFINX=0). SD (bit 31) is derived at read time. */ + inline void fp_state_dirty(); + + /* CV32E40P-only CSRs, absent from the generic register file. */ + Cv32e40pRoCsr mvendorid_ro; /* 0xF11 (replaces the base read/write reg) */ + Cv32e40pRoCsr marchid_ro; /* 0xF12 (replaces the base read/write reg) */ + Cv32e40pRoCsr mimpid; /* 0xF13 */ + Cv32e40pRoCsr mhartid_csr; /* 0xF14 */ + CsrReg tinfo; /* 0x7A4, read-only through a zero mask */ + CsrReg mcontext; /* 0x7A8, writable only from debug mode */ + CsrReg scontext; /* 0x7AA, writable only from debug mode */ + CsrReg minstret; /* 0xB02 */ +#if ISS_REG_WIDTH == 32 + CsrReg mcycleh; /* 0xB80 */ + CsrReg minstreth; /* 0xB82 */ +#endif + CsrReg mhpmevent[29]; /* 0x323..0x33F */ + + /* PULP custom CSRs (COREV_PULP configurations). */ + Cv32e40pRoCsr uhartid; /* 0xCD0 */ + Cv32e40pRoCsr privlv; /* 0xCD1 */ + Cv32e40pRoCsr zfinx_csr; /* 0xCD2, undeclared when FPU=1 && ZFINX=0 */ + + /* Hardware-loop CSRs: 0xCC0..0xCC2 / 0xCC4..0xCC6 (gap at 0xCC3). */ + Cv32e40pHwloopCsr hwloop_csr[6]; + + /* fflags / frm / fcsr (0x001..0x003). */ + Cv32e40pFpCsr fflags_csr; + Cv32e40pFpCsr frm_csr; + Cv32e40pFpCsr fcsr_csr; + +private: + bool fflags_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool frm_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool fcsr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); + bool tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool mstatus_read_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); + + int64_t mcycle_offset = 0; +}; + +inline bool Cv32e40pCsr::fp_access_illegal() +{ + /* RTL (cv32e40p_cs_registers.sv:1110 + decoder): illegal when there is + * no FPU; gated on mstatus.FS only with the FPU registers in the ISA; + * always legal for ZFINX (no FS state, flags/rm still implemented). */ +#if CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA + return this->mstatus.fs == 0; +#elif CONFIG_GVSOC_ISS_CV32E40P_ZFINX + return false; +#else + return true; +#endif +} + +inline void Cv32e40pCsr::fp_state_dirty() +{ +#if CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA + this->mstatus.fs = 3; +#endif +} diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp new file mode 100644 index 00000000..b30a086a --- /dev/null +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -0,0 +1,316 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +/* CV32E40P CSR personality for iss_v2. + * + * Same CSR map and write-legality rules as the v1 model + * (cpu/iss/src/cv32e40p/csr_cv32e40p.cpp), expressed as a Csr subclass: + * base registers are tightened through their write masks, core-only + * registers are declared here, and registers the core does not implement + * are undeclared so access falls through to the unsupported-CSR path, + * which this core configures to raise illegal-instruction. */ + +#include + +bool Cv32e40pRoCsr::check_access(Iss *iss, bool write, bool read) +{ + if (write) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return false; + } + return true; +} + +bool Cv32e40pFpCsr::check_access(Iss *iss, bool write, bool read) +{ + if (iss->csr.fp_access_illegal()) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return false; + } + return true; +} + +bool Cv32e40pHwloopCsr::check_access(Iss *iss, bool write, bool read) +{ + if (write) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return false; + } + return true; +} + +Cv32e40pCsr::Cv32e40pCsr(Iss &iss) +: Csr(iss) +{ + /* M-mode-only core: drop the registers the RTL does not implement so + * access raises illegal-instruction through the unsupported-CSR path. + * Compared with the v1 model this also drops satp and the user + * counters (cycle/time/instret): no U-mode, no mcounteren. */ + const iss_reg_t nonexistent[] = { + 0x100, 0x104, 0x105, 0x106, /* sstatus, sie, stvec, scounteren */ + 0x140, 0x141, 0x142, 0x143, 0x144, /* sscratch, sepc, scause, stval, sip */ + 0x180, /* satp */ + 0x302, 0x303, /* medeleg, mideleg */ + 0x306, /* mcounteren */ + 0x740, 0x741, 0x742, 0x744, /* mnscratch, mnepc, mncause, mnstatus */ + 0x008, 0x009, 0x00A, 0x00F, /* vstart, vxstat, vxrm, vcsr */ + 0xC20, 0xC21, 0xC22, /* vl, vtype, vlenb */ + 0xC00, 0xC01, 0xC02, /* cycle, time, instret */ + }; + for (iss_reg_t addr : nonexistent) + { + this->undeclare_csr(addr); + } + + this->raise_on_unsupported_csr = true; + + /* Machine information registers: read-only, writes raise illegal. + * mvendorid/marchid are re-declared with the read-only register type + * (the base class versions accept writes). */ + this->undeclare_csr(0xF11); + this->undeclare_csr(0xF12); + this->declare_csr(&this->mvendorid_ro, "mvendorid", 0xF11, 0x00000602); + this->declare_csr(&this->marchid_ro, "marchid", 0xF12, 0x00000004); +#if CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA || CONFIG_GVSOC_ISS_CV32E40P_ZFINX || CONFIG_GVSOC_ISS_CV32E40P_PULP + this->declare_csr(&this->mimpid, "mimpid", 0xF13, 1); +#else + this->declare_csr(&this->mimpid, "mimpid", 0xF13, 0); +#endif + this->declare_csr(&this->mhartid_csr, "mhartid", 0xF14, this->mhartid); + + /* Counter CSRs. */ + this->declare_csr(&this->minstret, "minstret", 0xB02); +#if ISS_REG_WIDTH == 32 + this->declare_csr(&this->mcycleh, "mcycleh", 0xB80); + this->declare_csr(&this->minstreth, "minstreth", 0xB82); +#endif + + /* mcycle: value is derived from the clock with a write offset, and + * freezes on mcountinhibit.CY. Registered after the base callback so + * this one has the last word. */ + this->mcycle.register_callback(std::bind(&Cv32e40pCsr::mcycle_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + + /* mcountinhibit: reset with all implemented bits set (RTL behaviour), + * i.e. counters disabled out of reset. */ + const int num_hpm = CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS; + iss_reg_t mcountinhibit_mask = 0x5 | (((1u << num_hpm) - 1) << 3); + this->mcountinhibit.set_write_mask(mcountinhibit_mask); + this->mcountinhibit.reset_val = mcountinhibit_mask; + + /* HPM counters and event selectors: only the first num_mhpmcounters + * are implemented, the rest are WARL zero (writes ignored). Only + * mhpmevent bits [15:0] exist (16 HPM event lines). */ + for (int i = 0; i < 29; i++) + { + iss_reg_t counter_mask = (i < num_hpm) ? (iss_reg_t)-1 : 0; + this->mhpmcounter[i].set_write_mask(counter_mask); +#if ISS_REG_WIDTH == 32 + this->mhpmcounterh[i].set_write_mask(counter_mask); +#endif + this->declare_csr(&this->mhpmevent[i], "mhpmevent" + std::to_string(i + 3), + 0x323 + i, 0, (i < num_hpm) ? 0xFFFF : 0); + } + + /* Interrupt and trap CSRs: masks from the RTL (cv32e40p_cs_registers.sv). + * mstatus: only MIE/MPIE (and FS with an FPU) are writable; the mask is + * applied by Core::mstatus_update via CONFIG_GVSOC_ISS_CORE_MSTATUS_WRITE_MASK + * set by the recipe. Reset is MPP=M, FS=Off for every configuration. */ + this->mstatus.reset_val = 0x00001800; + this->mie.set_write_mask(0xFFFF0888); + this->mip.set_write_mask(0); + this->mtvec.set_write_mask(0xFFFFFF01); + this->mtvec.reset_val = 0x1; + this->mtval.set_write_mask(0); + this->mcause.set_write_mask(0x8000001F); + + /* Trigger module: one trigger, tselect hardwired to 0, tdata* writable + * only from debug mode (not modelled), tinfo reports type 2. */ + this->tselect.set_write_mask(0); + this->tselect.register_callback(std::bind(&Cv32e40pCsr::tselect_read_zero, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->tdata1.reset_val = 0x28001040; + this->tdata1.set_write_mask(0); + this->tdata2.set_write_mask(0); + this->tdata3.set_write_mask(0); + this->declare_csr(&this->tinfo, "tinfo", 0x7A4, 0x4, 0); + this->declare_csr(&this->mcontext, "mcontext", 0x7A8, 0, 0); + this->declare_csr(&this->scontext, "scontext", 0x7AA, 0, 0); + +#if CONFIG_GVSOC_ISS_CV32E40P_PULP + /* PULP custom CSRs. The RTL decoder raises illegal-instruction on any + * write to them (read-only register type). */ + this->declare_csr(&this->uhartid, "uhartid", 0xCD0, this->mhartid); + this->declare_csr(&this->privlv, "privlv", 0xCD1, 3); +#if !CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA + /* zfinx indicator: reads 1 on ZFINX, 0 without an FPU. With FPU=1 && + * ZFINX=0 the RTL rejects even reads, so it stays undeclared and the + * unsupported-CSR path raises illegal-instruction. */ + this->declare_csr(&this->zfinx_csr, "zfinx", 0xCD2, + CONFIG_GVSOC_ISS_CV32E40P_ZFINX ? 1 : 0); +#endif + + /* Hardware-loop CSRs, readable via CSR instructions only (writes go + * through the cv.* instructions and csrrw raises illegal). The values + * live in the Hwloop module. */ + for (int loop = 0; loop < 2; loop++) + { + static const char *names[] = { "lpstart", "lpend", "lpcount" }; + for (int kind = 0; kind < 3; kind++) + { + int index = loop * 3 + kind; + this->declare_csr(&this->hwloop_csr[index], + names[kind] + std::to_string(loop), 0xCC0 + loop * 4 + kind); + this->hwloop_csr[index].register_callback( + std::bind(&Cv32e40pCsr::hwloop_csr_access, this, + std::placeholders::_1, std::placeholders::_2, + std::placeholders::_3, index)); + } + } +#endif + + /* fflags / frm / fcsr front-ends: legality gated on mstatus.FS by the + * register type, value mapping onto the shared fcsr field here. */ + this->declare_csr(&this->fflags_csr, "fflags", 0x001); + this->fflags_csr.register_callback(std::bind(&Cv32e40pCsr::fflags_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->frm_csr, "frm", 0x002); + this->frm_csr.register_callback(std::bind(&Cv32e40pCsr::frm_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->fcsr_csr, "fcsr", 0x003); + this->fcsr_csr.register_callback(std::bind(&Cv32e40pCsr::fcsr_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +} + +void Cv32e40pCsr::start() +{ + /* Registered here so it runs after Core::mstatus_update, which is + * registered by the Core constructor (after this class is built) and + * overwrites the read value with the stored one. */ + this->mstatus.register_callback(std::bind(&Cv32e40pCsr::mstatus_read_fixup, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +} + +void Cv32e40pCsr::reset(bool active) +{ + Csr::reset(active); + + if (active) + { + /* dcsr: xdebugver=4 in [31:28], prv=M in [1:0] (base reset leaves + * prv=0). */ + this->dcsr = (4 << 28) | 0x3; + + this->mcycle_offset = 0; + } +} + +bool Cv32e40pCsr::mstatus_read_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ +#if CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA + /* SD (bit 31) is derived on read: set when FS or XS is Dirty. */ + if (!is_write) + { + if (((value >> 13) & 3) == 3 || ((value >> 15) & 3) == 3) + { + value |= 1ULL << 31; + } + } +#endif + return false; +} + +bool Cv32e40pCsr::tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + /* One trigger: tselect always reads 0 (the base callback reads -1). */ + if (!is_write) + { + value = 0; + } + return false; +} + +bool Cv32e40pCsr::mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->mcycle.value = value; + this->mcycle_offset = (int64_t)value - (int64_t)this->iss.clock.get_cycles(); + } + else + { + if (this->mcountinhibit.value & 0x1) + { + value = this->mcycle.value; + } + else + { + value = (iss_reg_t)((int64_t)this->iss.clock.get_cycles() + this->mcycle_offset); + } + } + return false; +} + +bool Cv32e40pCsr::hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index) +{ + int loop = index / 3; + + switch (index % 3) + { + case 0: value = this->iss.hwloop.get_start(loop); break; + /* The Hwloop module stores the loop-back point, LPEND - 4 (see the + * corev.hpp setters); the architectural LPEND is re-derived here. */ + case 1: value = this->iss.hwloop.get_end(loop) + 4; break; + case 2: value = this->iss.hwloop.get_count(loop); break; + } + return false; +} + +bool Cv32e40pCsr::fflags_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->fcsr.fflags = value; + /* RTL: an fflags write (fflags_we_i) forces mstatus.FS=Dirty. */ + this->fp_state_dirty(); + } + else + { + value = this->fcsr.fflags; + } + return false; +} + +bool Cv32e40pCsr::frm_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->fcsr.frm = value; + this->fp_state_dirty(); + } + else + { + value = this->fcsr.frm; + } + return false; +} + +bool Cv32e40pCsr::fcsr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->fcsr.raw = value & 0xff; + this->fp_state_dirty(); + } + else + { + value = this->fcsr.raw; + } + return false; +} diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index 156bab16..e4ce9f58 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -21,9 +21,10 @@ from __future__ import annotations from typing import Iterable +from typing_extensions import override from gvsoc.systree import Component -from cpu.iss_v2.riscv import (RiscvCommon, IrqExternal, ExecInOrder, Regfile, - Csr, Event, LsuV2, Hwloop) +from cpu.iss_v2.riscv import (RiscvCommon, IssModule, Irq, ExecInOrder, + Regfile, Event, LsuV2, Hwloop) from cpu.iss.isa_gen.isa_gen import Isa, IsaSubset from cpu.iss.isa_gen.isa_riscv_gen import RiscvIsa from cpu.iss.isa_gen.isa_cv32e40pv2 import CoreV2 @@ -31,11 +32,49 @@ isa_instances: dict[tuple[str, str], Isa] = {} +# misa: MXL=1 | I | M | C, plus X for the PULP extensions and F when the +# FPU registers are in the ISA (not for ZFINX). Same values as the v1 +# model (pulp/cpu/iss/pulp_cores.py). +_MISA_BASE = 0x40001104 + class Cv32e40pConfig(RiscvConfig): pass +class Cv32e40pCsr(IssModule): + """CV32E40P CSR personality. + + Selects the Cv32e40pCsr C++ class for the csr slot: M-mode-only CSR + map, RTL write masks, PULP custom CSRs, hardware-loop CSRs and + illegal-instruction on unsupported CSR accesses. + """ + + def __init__(self, fpu: bool=False, zfinx: bool=False, pulp: bool=True, + num_mhpmcounters: int=1): + self.fpu = fpu + self.zfinx = zfinx + self.pulp = pulp + self.num_mhpmcounters = num_mhpmcounters + + @override + def gen(self, iss: RiscvCommon): + iss.isa.add_define('CONFIG_GVSOC_ISS_CSR', 'Cv32e40pCsr') + iss.isa.add_include('') + iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA', 1 if self.fpu else 0) + iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_ZFINX', 1 if self.zfinx else 0) + iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_PULP', 1 if self.pulp else 0) + iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS', self.num_mhpmcounters) + # mstatus write policy, applied by Core::mstatus_update: only + # MIE/MPIE are writable, plus FS with FPU registers in the ISA. + iss.isa.add_define('CONFIG_GVSOC_ISS_CORE_MSTATUS_WRITE_MASK', + '0x6088' if self.fpu else '0x88') + iss.add_sources([ + 'cpu/iss_v2/src/cores/cv32e40p/csr.cpp', + 'cpu/iss_v2/src/csr.cpp', + ]) + + class Cv32e40p(RiscvCommon): """CV32E40P on the iss_v2 modular core. @@ -49,26 +88,42 @@ class Cv32e40p(RiscvCommon): isa_name: str = 'cv32e40p_v2' def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, + fpu: bool=False, zfinx: bool=False, pulp: bool=True, + num_mhpmcounters: int=1, extra_extensions: Iterable[IsaSubset] = ()): - cache_key = (type(self).isa_name, config.isa) + # The PULP extensions change the decoded instruction set, so they + # are part of the cache identity and of the generated ISA name. + isa_tag = f"{config.isa}_pulp" if pulp else config.isa + cache_key = (type(self).isa_name, isa_tag) isa_instance: Isa | None = isa_instances.get(cache_key) if isa_instance is None: extensions: list[IsaSubset] = [ *extra_extensions, - CoreV2(), ] + if pulp: + extensions.append(CoreV2()) - isa_instance = RiscvIsa(f"{type(self).isa_name}_{config.isa}", + isa_instance = RiscvIsa(f"{type(self).isa_name}_{isa_tag}", config.isa, extensions=extensions) isa_instances[cache_key] = isa_instance - modules: dict[str, object] = { - 'irq': IrqExternal(), + misa = _MISA_BASE + if fpu and not zfinx: + misa |= 1 << 5 # F + if pulp: + misa |= 1 << 23 # X + + modules: dict[str, IssModule] = { + # RISC-V privileged interrupt/exception scheme (mie/mip/mtvec, + # standard mcause codes), as implemented by the RTL. The PULP + # event-unit style IrqExternal does not apply to this core. + 'irq': Irq(), 'event': Event(), - 'csr': Csr(), + 'csr': Cv32e40pCsr(fpu=fpu, zfinx=zfinx, pulp=pulp, + num_mhpmcounters=num_mhpmcounters), 'exec': ExecInOrder(scoreboard=True, inorder_commit=True), 'lsu': LsuV2(), 'regfile': Regfile(scoreboard=True), @@ -76,4 +131,4 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, } super().__init__(parent, name, config=config, isa=isa_instance, - modules=modules) + misa=misa, modules=modules) From 9e515294fd09cbc9626beef2e93ae4fdbcb0ebe3 Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 17 Jul 2026 11:40:52 +0200 Subject: [PATCH 06/28] feat(cv32e40p): FPU and ZFINX configurations on the iss_v2 bring-up targets The recipe forwards zfinx to the v2 core (register-file routing), tags the ISA cache and the generated ISA name with the pulp/zfinx variant, and emits CONFIG_GVSOC_ISS_FP_STATE_DIRTY on FPU configurations so FP write-backs dirty mstatus.FS as in the RTL. The bring-up platform moves to pulp/cv32e40p_v2_spike.py, shared by three thin targets (cv32e40p-v2-spike, -fpu, -zfinx): one target name per core configuration, so each gets its own serialized platform tree (a non-default --parameter run would fall back to the JSON config path, which the io_v2 components do not support). --- cv32e40p-v2-spike-fpu.py | 37 +++++++++ cv32e40p-v2-spike-zfinx.py | 38 +++++++++ cv32e40p-v2-spike.py | 119 ++-------------------------- pulp/cpu/iss/cv32e40p_v2.py | 11 ++- pulp/cv32e40p_v2_spike.py | 152 ++++++++++++++++++++++++++++++++++++ 5 files changed, 240 insertions(+), 117 deletions(-) create mode 100644 cv32e40p-v2-spike-fpu.py create mode 100644 cv32e40p-v2-spike-zfinx.py create mode 100644 pulp/cv32e40p_v2_spike.py diff --git a/cv32e40p-v2-spike-fpu.py b/cv32e40p-v2-spike-fpu.py new file mode 100644 index 00000000..9e6124ea --- /dev/null +++ b/cv32e40p-v2-spike-fpu.py @@ -0,0 +1,37 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 bring-up target, FPU configuration (rv32imf + CoreV). + +import gvsoc.runner +from pulp.cv32e40p_v2_spike import Cv32e40pSpikeTop + + +class Cv32e40p(Cv32e40pSpikeTop): + + def __init__(self, parent, name=None): + super().__init__(parent, name, fpu=1) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 bring-up (FPU)" + model = Cv32e40p + name = "cv32e40p-v2-spike-fpu" diff --git a/cv32e40p-v2-spike-zfinx.py b/cv32e40p-v2-spike-zfinx.py new file mode 100644 index 00000000..91068747 --- /dev/null +++ b/cv32e40p-v2-spike-zfinx.py @@ -0,0 +1,38 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 bring-up target, ZFINX configuration (FP on the integer +# register file). + +import gvsoc.runner +from pulp.cv32e40p_v2_spike import Cv32e40pSpikeTop + + +class Cv32e40p(Cv32e40pSpikeTop): + + def __init__(self, parent, name=None): + super().__init__(parent, name, zfinx=1) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 bring-up (ZFINX)" + model = Cv32e40p + name = "cv32e40p-v2-spike-zfinx" diff --git a/cv32e40p-v2-spike.py b/cv32e40p-v2-spike.py index 03d4f6d1..8aadbd22 100644 --- a/cv32e40p-v2-spike.py +++ b/cv32e40p-v2-spike.py @@ -18,128 +18,19 @@ # Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) # -# CV32E40P iss_v2 bring-up target. -# -# The iss_v2 LSU (LsuV2) drives the fetch/data ports with the io_v2 -# protocol, so the whole platform lives on the io_v2 plane: router_v2, -# memory_v3 and loader_v2, following the Ri5ky testbench layout -# (pulp/ri5ky/ri5ky_testbench.py). The MMIO peripheral is reused as-is -# from that testbench: putchar @ +0x0, exit @ +0x4. -# -# Integer-only for the bring-up: rv32imc + the CoreV (XPULP v2) subset. +# CV32E40P iss_v2 bring-up target, integer configuration (rv32im + CoreV). +# Platform in pulp/cv32e40p_v2_spike.py, shared by the cv32e40p-v2-spike* +# variants. -import vp.clock_domain -import gvsoc.systree import gvsoc.runner -from gvrun.parameter import TargetParameter -from config_tree import Config, cfg_field -from memory.memory_v3 import Memory, MemoryV3Config -from interco.router_v2 import Router, RouterConfig, RouterMapping -from utils.loader.loader_v2 import ElfLoader -from pulp.ri5ky.ri5ky_mmio import Ri5kyMmio -from pulp.cpu.iss.cv32e40p_v2 import Cv32e40p as Cv32e40pV2Core, Cv32e40pConfig - - -class Cv32e40pSpikeConfig(Config): - """Configuration for the CV32E40P iss_v2 bring-up SoC. - - Minimal layout: - - mem at 0x0000_0000 (4 MB), entry point 0x80 as in the RTL testbench - - MMIO at 0x1000_0000 (4 KB): putchar @ +0, exit @ +4 - """ - - mem_base: int = cfg_field(default=0x0000_0000, fmt="hex", dump=True, desc=( - "Base address of the main memory" - )) - - mem_size: int = cfg_field(default=0x40_0000, fmt="hex", dump=True, desc=( - "Size of the main memory" - )) - - mmio_base: int = cfg_field(default=0x1000_0000, fmt="hex", dump=True, desc=( - "Base address of the MMIO peripheral" - )) - - mmio_size: int = cfg_field(default=0x1000, fmt="hex", dump=True, desc=( - "Size of the MMIO peripheral window" - )) - - boot_addr: int = cfg_field(default=0x80, fmt="hex", dump=True, desc=( - "Boot address (matches RTL BOOT_ADDR)" - )) - - core: Cv32e40pConfig = cfg_field(init=False, desc=( - "CV32E40P core configuration" - )) - - mem: MemoryV3Config = cfg_field(init=False, desc=( - "Backing memory configuration" - )) - - router: RouterConfig = cfg_field(init=False, desc=( - "Router configuration" - )) +from pulp.cv32e40p_v2_spike import Cv32e40pSpikeTop - mem_mapping: RouterMapping = cfg_field(init=False, desc=( - "Address range of the main memory" - )) - mmio_mapping: RouterMapping = cfg_field(init=False, desc=( - "Address range of the MMIO peripheral" - )) - - def __post_init__(self): - super().__post_init__() - # No compressed for now: with C enabled the generated decode table - # references the rvf c.flwsp/c.fswsp handlers even when they are - # inactive (no F), and without an FPU module nothing declares them. - self.core = Cv32e40pConfig(isa='rv32im', boot_addr=self.boot_addr) - self.mem = MemoryV3Config('mem', size=self.mem_size, atomics=False, latency=1) - self.router = RouterConfig(kind='bandwidth') - self.mem_mapping = RouterMapping(name='mem_mapping', - base=self.mem_base, size=self.mem_size) - self.mmio_mapping = RouterMapping(name='mmio_mapping', - base=self.mmio_base, size=self.mmio_size) - - -class Cv32e40pSpikeSoc(gvsoc.systree.Component): - - def __init__(self, parent, name, config: Cv32e40pSpikeConfig, binary): - super().__init__(parent, name, config=config) - - mem = Memory ( self, 'mem' , config=config.mem ) - mmio = Ri5kyMmio ( self, 'mmio' ) - ico = Router ( self, 'ico' , config=config.router ) - core = Cv32e40pV2Core( self, 'core' , config=config.core ) - loader = ElfLoader ( self, 'loader', binary=binary ) - - ico.o_MAP ( mem.i_INPUT() , mapping=config.mem_mapping ) - ico.o_MAP ( mmio.i_INPUT(), mapping=config.mmio_mapping ) - - # Three independent masters, one router input port each. - loader.o_OUT ( ico.i_INPUT(0) ) - loader.o_START ( core.i_FETCHEN() ) - loader.o_ENTRY ( core.i_ENTRY() ) - - core.o_FETCH ( ico.i_INPUT(1) ) - core.o_DATA ( ico.i_INPUT(2) ) - - -class Cv32e40p(gvsoc.systree.Component): +class Cv32e40p(Cv32e40pSpikeTop): def __init__(self, parent, name=None): super().__init__(parent, name) - binary = TargetParameter( - self, name='binary', value=None, description='ELF binary to simulate' - ).get_value() - - config = Cv32e40pSpikeConfig('soc') - - clock = vp.clock_domain.Clock_domain(self, 'clock', frequency=50000000) - soc = Cv32e40pSpikeSoc(self, 'soc', config, binary) - clock.o_CLOCK(soc.i_CLOCK()) - class Target(gvsoc.runner.Target): diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index e4ce9f58..8b5738ca 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -62,6 +62,9 @@ def gen(self, iss: RiscvCommon): iss.isa.add_define('CONFIG_GVSOC_ISS_CSR', 'Cv32e40pCsr') iss.isa.add_include('') iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA', 1 if self.fpu else 0) + if self.fpu: + # FP write-backs must dirty mstatus.FS (see iss_v2 isa_lib/macros.h). + iss.isa.add_define('CONFIG_GVSOC_ISS_FP_STATE_DIRTY', 1) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_ZFINX', 1 if self.zfinx else 0) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_PULP', 1 if self.pulp else 0) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS', self.num_mhpmcounters) @@ -92,9 +95,11 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, num_mhpmcounters: int=1, extra_extensions: Iterable[IsaSubset] = ()): - # The PULP extensions change the decoded instruction set, so they - # are part of the cache identity and of the generated ISA name. + # pulp and zfinx change what gets compiled behind one ISA string, + # so both are part of the cache key and of the generated ISA name. isa_tag = f"{config.isa}_pulp" if pulp else config.isa + if zfinx: + isa_tag += '_zfinx' cache_key = (type(self).isa_name, isa_tag) isa_instance: Isa | None = isa_instances.get(cache_key) @@ -131,4 +136,4 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, } super().__init__(parent, name, config=config, isa=isa_instance, - misa=misa, modules=modules) + misa=misa, zfinx=zfinx, modules=modules) diff --git a/pulp/cv32e40p_v2_spike.py b/pulp/cv32e40p_v2_spike.py new file mode 100644 index 00000000..cb143ec2 --- /dev/null +++ b/pulp/cv32e40p_v2_spike.py @@ -0,0 +1,152 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 bring-up platform, shared by the cv32e40p-v2-spike* +# targets (one target name per core configuration, so each gets its own +# serialized platform tree). +# +# The iss_v2 LSU (LsuV2) drives the fetch/data ports with the io_v2 +# protocol, so the whole platform lives on the io_v2 plane: router_v2, +# memory_v3 and loader_v2, following the Ri5ky testbench layout +# (pulp/ri5ky/ri5ky_testbench.py). The MMIO peripheral is reused as-is +# from that testbench: putchar @ +0x0, exit @ +0x4. + +import vp.clock_domain +import gvsoc.systree +from gvrun.parameter import TargetParameter +from config_tree import Config, cfg_field +from memory.memory_v3 import Memory, MemoryV3Config +from interco.router_v2 import Router, RouterConfig, RouterMapping +from utils.loader.loader_v2 import ElfLoader +from pulp.ri5ky.ri5ky_mmio import Ri5kyMmio +from pulp.cpu.iss.cv32e40p_v2 import Cv32e40p as Cv32e40pV2Core, Cv32e40pConfig + + +class Cv32e40pSpikeConfig(Config): + """Configuration for the CV32E40P iss_v2 bring-up SoC. + + Minimal layout: + - mem at 0x0000_0000 (4 MB), entry point 0x80 as in the RTL testbench + - MMIO at 0x1000_0000 (4 KB): putchar @ +0, exit @ +4 + """ + + mem_base: int = cfg_field(default=0x0000_0000, fmt="hex", dump=True, desc=( + "Base address of the main memory" + )) + + mem_size: int = cfg_field(default=0x40_0000, fmt="hex", dump=True, desc=( + "Size of the main memory" + )) + + mmio_base: int = cfg_field(default=0x1000_0000, fmt="hex", dump=True, desc=( + "Base address of the MMIO peripheral" + )) + + mmio_size: int = cfg_field(default=0x1000, fmt="hex", dump=True, desc=( + "Size of the MMIO peripheral window" + )) + + boot_addr: int = cfg_field(default=0x80, fmt="hex", dump=True, desc=( + "Boot address (matches RTL BOOT_ADDR)" + )) + + fpu: int = cfg_field(default=0, dump=True, desc=( + "FPU configuration (F extension, FP register file)" + )) + + zfinx: int = cfg_field(default=0, dump=True, desc=( + "ZFINX configuration (FP operations on the integer register file)" + )) + + core: Cv32e40pConfig = cfg_field(init=False, desc=( + "CV32E40P core configuration" + )) + + mem: MemoryV3Config = cfg_field(init=False, desc=( + "Backing memory configuration" + )) + + router: RouterConfig = cfg_field(init=False, desc=( + "Router configuration" + )) + + mem_mapping: RouterMapping = cfg_field(init=False, desc=( + "Address range of the main memory" + )) + + mmio_mapping: RouterMapping = cfg_field(init=False, desc=( + "Address range of the MMIO peripheral" + )) + + def __post_init__(self): + super().__post_init__() + # No compressed for now: with C enabled the generated decode table + # references the rvf c.flwsp/c.fswsp handlers even when they are + # inactive (no F), and without an FPU module nothing declares them. + # ZFINX also needs the F opcodes in the decoder (routed to the + # integer register file). + isa = 'rv32imf' if (self.fpu or self.zfinx) else 'rv32im' + self.core = Cv32e40pConfig(isa=isa, boot_addr=self.boot_addr) + self.mem = MemoryV3Config('mem', size=self.mem_size, atomics=False, latency=1) + self.router = RouterConfig(kind='bandwidth') + self.mem_mapping = RouterMapping(name='mem_mapping', + base=self.mem_base, size=self.mem_size) + self.mmio_mapping = RouterMapping(name='mmio_mapping', + base=self.mmio_base, size=self.mmio_size) + + +class Cv32e40pSpikeSoc(gvsoc.systree.Component): + + def __init__(self, parent, name, config: Cv32e40pSpikeConfig, binary): + super().__init__(parent, name, config=config) + + mem = Memory ( self, 'mem' , config=config.mem ) + mmio = Ri5kyMmio ( self, 'mmio' ) + ico = Router ( self, 'ico' , config=config.router ) + core = Cv32e40pV2Core( self, 'core' , config=config.core , + fpu=bool(config.fpu), zfinx=bool(config.zfinx) ) + loader = ElfLoader ( self, 'loader', binary=binary ) + + ico.o_MAP ( mem.i_INPUT() , mapping=config.mem_mapping ) + ico.o_MAP ( mmio.i_INPUT(), mapping=config.mmio_mapping ) + + # Three independent masters, one router input port each. + loader.o_OUT ( ico.i_INPUT(0) ) + loader.o_START ( core.i_FETCHEN() ) + loader.o_ENTRY ( core.i_ENTRY() ) + + core.o_FETCH ( ico.i_INPUT(1) ) + core.o_DATA ( ico.i_INPUT(2) ) + + +class Cv32e40pSpikeTop(gvsoc.systree.Component): + + def __init__(self, parent, name=None, fpu: int=0, zfinx: int=0): + super().__init__(parent, name) + + binary = TargetParameter( + self, name='binary', value=None, description='ELF binary to simulate' + ).get_value() + + config = Cv32e40pSpikeConfig('soc', fpu=fpu, zfinx=zfinx) + + clock = vp.clock_domain.Clock_domain(self, 'clock', frequency=50000000) + soc = Cv32e40pSpikeSoc(self, 'soc', config, binary) + clock.o_CLOCK(soc.i_CLOCK()) From ac97e27e88e1b3fd43618fb86ba757504384c26e Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 17 Jul 2026 12:43:07 +0200 Subject: [PATCH 07/28] feat(cv32e40p): compressed ISA on the v2 bring-up targets The spike platform now builds rv32imc (base) and rv32imfc (fpu/zfinx) decode tables. On the ZFINX variant the recipe disables the compressed FP loads/stores: the RTL compressed decoder only accepts them with FPU == 1 && ZFINX == 0 (cv32e40p_compressed_decoder.sv). --- pulp/cpu/iss/cv32e40p_v2.py | 5 +++++ pulp/cv32e40p_v2_spike.py | 10 ++++------ 2 files changed, 9 insertions(+), 6 deletions(-) diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index 8b5738ca..38d05e39 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -113,6 +113,11 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, isa_instance = RiscvIsa(f"{type(self).isa_name}_{isa_tag}", config.isa, extensions=extensions) + if zfinx: + # RTL decodes the compressed FP loads/stores only with + # FPU == 1 && ZFINX == 0 (cv32e40p_compressed_decoder.sv). + isa_instance.disable_from_isa_tag('cf') + isa_instances[cache_key] = isa_instance misa = _MISA_BASE diff --git a/pulp/cv32e40p_v2_spike.py b/pulp/cv32e40p_v2_spike.py index cb143ec2..7dd72e0b 100644 --- a/pulp/cv32e40p_v2_spike.py +++ b/pulp/cv32e40p_v2_spike.py @@ -97,12 +97,10 @@ class Cv32e40pSpikeConfig(Config): def __post_init__(self): super().__post_init__() - # No compressed for now: with C enabled the generated decode table - # references the rvf c.flwsp/c.fswsp handlers even when they are - # inactive (no F), and without an FPU module nothing declares them. - # ZFINX also needs the F opcodes in the decoder (routed to the - # integer register file). - isa = 'rv32imf' if (self.fpu or self.zfinx) else 'rv32im' + # ZFINX needs the F opcodes in the decoder (routed to the integer + # register file); the compressed FP rows are disabled by the core + # recipe to match the RTL compressed decoder. + isa = 'rv32imfc' if (self.fpu or self.zfinx) else 'rv32imc' self.core = Cv32e40pConfig(isa=isa, boot_addr=self.boot_addr) self.mem = MemoryV3Config('mem', size=self.mem_size, atomics=False, latency=1) self.router = RouterConfig(kind='bandwidth') From 41b6eded9705ebd5cf68e76fc5f735e7be821f9b Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 17 Jul 2026 14:33:32 +0200 Subject: [PATCH 08/28] feat(cv32e40p): performance-counter personality on the v2 core Counter model matching the RTL mhpm scheme (cv32e40p_cs_registers.sv): minstret counts retires gated on mcountinhibit.IR, mhpmcounterN advances at most once per retire when its mhpmeventN mask intersects the event lines the instruction fired, mcycle/mcycleh become one 64-bit count that freezes at the current value while mcountinhibit.CY is set. Events are accumulated per instruction by Cv32e40pEvents and committed at retire (trapping instructions drop them); Cv32e40pExec keeps the core on the full handlers while any counter is enabled, since only those fire the event lines (same scheme as the Ri5ky PCMR pairing). Only the architectural lines are modelled: instr, load, store, jump, branch, branch taken, compressed. The timing lines (stalls, imiss, APU) never fire and their counters read zero. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 59 +++++++++++ cpu/iss_v2/include/cores/cv32e40p/events.hpp | 44 +++++++++ .../include/cores/cv32e40p/events_implem.hpp | 59 +++++++++++ cpu/iss_v2/include/cores/cv32e40p/exec.hpp | 19 ++++ .../include/cores/cv32e40p/exec_implem.hpp | 19 ++++ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 99 ++++++++++++++++--- cpu/iss_v2/src/cores/cv32e40p/events.cpp | 16 +++ pulp/cpu/iss/cv32e40p_v2.py | 41 +++++++- 8 files changed, 337 insertions(+), 19 deletions(-) create mode 100644 cpu/iss_v2/include/cores/cv32e40p/events.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/exec.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp create mode 100644 cpu/iss_v2/src/cores/cv32e40p/events.cpp diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index 0332cadc..b4eacb89 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -28,6 +28,8 @@ #ifndef CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS #define CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS 1 #endif +static_assert(CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS <= 29, + "at most 29 HPM counters (mhpmcounter3..31)"); /* CSR that is read-only from CSR instructions: any write attempt raises an * illegal-instruction exception before the access happens, so the destination @@ -62,6 +64,10 @@ class Cv32e40pHwloopCsr : public CsrAbtractReg class Cv32e40pCsr : public Csr { public: + /* mcountinhibit implemented bits: CY, IR and one per HPM counter. */ + static constexpr iss_reg_t MCOUNTINHIBIT_MASK = + 0x5 | (((1u << CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS) - 1) << 3); + Cv32e40pCsr(Iss &iss); void start(); @@ -75,6 +81,15 @@ class Cv32e40pCsr : public Csr * in the ISA (FPU=1, ZFINX=0). SD (bit 31) is derived at read time. */ inline void fp_state_dirty(); + /* Advance the counters for one retired instruction: events is the OR of + * the RTL hpm_events lines it fired (see cores/cv32e40p/events.hpp). + * Called once per retire by Cv32e40pEvents::event_retire_account. */ + inline void hpm_commit(uint32_t events); + + /* True while any implemented counter is enabled: keeps the core on the + * full handlers, where the event lines fire (Cv32e40pExec). */ + inline bool hpm_counting(); + /* CV32E40P-only CSRs, absent from the generic register file. */ Cv32e40pRoCsr mvendorid_ro; /* 0xF11 (replaces the base read/write reg) */ Cv32e40pRoCsr marchid_ro; /* 0xF12 (replaces the base read/write reg) */ @@ -110,8 +125,15 @@ class Cv32e40pCsr : public Csr bool hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); bool tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool mcountinhibit_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mstatus_read_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); + /* Current 64-bit mcycle count: the frozen register pair while + * mcountinhibit.CY is set, the offset clock otherwise. */ + uint64_t mcycle_count(); + void mcycle_set(uint64_t count); + int64_t mcycle_offset = 0; }; @@ -135,3 +157,40 @@ inline void Cv32e40pCsr::fp_state_dirty() this->mstatus.fs = 3; #endif } + +inline bool Cv32e40pCsr::hpm_counting() +{ + /* CY excluded: mcycle is clock-derived and needs no full-handler + * support, only minstret and the event counters do. */ + constexpr iss_reg_t event_bits = MCOUNTINHIBIT_MASK & ~(iss_reg_t)0x1; + return (this->mcountinhibit.value & event_bits) != event_bits; +} + +inline void Cv32e40pCsr::hpm_commit(uint32_t events) +{ + /* minstret: retired instructions, gated on mcountinhibit.IR (bit 2). */ + if (!(this->mcountinhibit.value & 0x4)) + { + if (++this->minstret.value == 0) + { +#if ISS_REG_WIDTH == 32 + this->minstreth.value++; +#endif + } + } + /* mhpmcounterN advances at most +1 per retire when the mhpmeventN mask + * intersects the fired lines and its mcountinhibit bit is clear. */ + for (int i = 0; i < CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS; i++) + { + if ((this->mhpmevent[i].value & events) + && !(this->mcountinhibit.value & (1u << (3 + i)))) + { + if (++this->mhpmcounter[i].value == 0) + { +#if ISS_REG_WIDTH == 32 + this->mhpmcounterh[i].value++; +#endif + } + } + } +} diff --git a/cpu/iss_v2/include/cores/cv32e40p/events.hpp b/cpu/iss_v2/include/cores/cv32e40p/events.hpp new file mode 100644 index 00000000..863dab03 --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/events.hpp @@ -0,0 +1,44 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include +#include + +/* RTL hpm_events bit positions (cv32e40p_cs_registers.sv:1327). Only the + * architectural, instruction-derived lines are accounted by the model; the + * timing lines (cycle, stalls, imiss, APU) never fire. */ +#define CV32E40P_HPM_INSTR (1u << 1) +#define CV32E40P_HPM_LD (1u << 5) +#define CV32E40P_HPM_ST (1u << 6) +#define CV32E40P_HPM_JUMP (1u << 7) +#define CV32E40P_HPM_BRANCH (1u << 8) +#define CV32E40P_HPM_BRANCH_TAKEN (1u << 9) +#define CV32E40P_HPM_COMP_INSTR (1u << 10) + +class Cv32e40pEvents : public Events +{ +public: + Cv32e40pEvents(Iss &iss) : Events(iss) {} + + void reset(bool active); + + inline void event_load_account(int incr); + inline void event_store_account(int incr); + inline void event_branch_account(); + inline void event_taken_branch_account(); + inline void event_jump_account(); + inline void event_retire_account(iss_insn_t *insn); + +private: + /* Event lines fired by the executing instruction, committed as one OR + * mask at retire: each counter advances at most +1 per instruction, as + * the RTL advances at most +1 per cycle (cv32e40p_cs_registers.sv:1437). + * Relies on the LSU firing its hooks only for accepted requests (no + * hook before a stall is detected), so nothing leaks across retries. */ + uint32_t pending_events = 0; +}; diff --git a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp new file mode 100644 index 00000000..751b4b8c --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp @@ -0,0 +1,59 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include "cpu/iss_v2/include/cores/cv32e40p/csr.hpp" +#include +#include +#include + +inline void Cv32e40pEvents::event_load_account(int incr) +{ + Events::event_load_account(incr); + this->pending_events |= CV32E40P_HPM_LD; +} + +inline void Cv32e40pEvents::event_store_account(int incr) +{ + Events::event_store_account(incr); + this->pending_events |= CV32E40P_HPM_ST; +} + +inline void Cv32e40pEvents::event_branch_account() +{ + Events::event_branch_account(); + this->pending_events |= CV32E40P_HPM_BRANCH; +} + +inline void Cv32e40pEvents::event_taken_branch_account() +{ + /* A taken branch fires both RTL event lines (base cascades to + * event_branch_account, the explicit OR keeps that non-load-bearing). */ + Events::event_taken_branch_account(); + this->pending_events |= CV32E40P_HPM_BRANCH | CV32E40P_HPM_BRANCH_TAKEN; +} + +inline void Cv32e40pEvents::event_jump_account() +{ + Events::event_jump_account(); + this->pending_events |= CV32E40P_HPM_JUMP; +} + +inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) +{ + Events::event_retire_account(insn); + /* A trapping instruction does not retire: drop its event lines. */ + if (this->iss.exec.has_exception) + { + this->pending_events = 0; + return; + } + uint32_t events = this->pending_events | CV32E40P_HPM_INSTR + | (insn->size == 2 ? CV32E40P_HPM_COMP_INSTR : 0); + this->pending_events = 0; + this->iss.csr.hpm_commit(events); +} diff --git a/cpu/iss_v2/include/cores/cv32e40p/exec.hpp b/cpu/iss_v2/include/cores/cv32e40p/exec.hpp new file mode 100644 index 00000000..e76c5bb3 --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/exec.hpp @@ -0,0 +1,19 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include + +class Iss; + +class Cv32e40pExec : public ExecInOrder +{ +public: + Cv32e40pExec(Iss &iss) : ExecInOrder(iss) {} + + inline bool can_switch_to_fast_mode(); +}; diff --git a/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp new file mode 100644 index 00000000..288d6809 --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp @@ -0,0 +1,19 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include "cpu/iss_v2/include/cores/cv32e40p/exec.hpp" +#include "cpu/iss_v2/include/exec/exec_inorder.hpp" + +inline bool Cv32e40pExec::can_switch_to_fast_mode() +{ + if (!ExecInOrder::can_switch_to_fast_mode()) return false; + + /* The event lines only fire from the full handlers: stay there while + * any implemented counter is enabled (see Cv32e40pCsr::hpm_counting). */ + return !this->iss.csr.hpm_counting(); +} diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index b30a086a..d70fe499 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -91,18 +91,24 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) this->declare_csr(&this->minstreth, "minstreth", 0xB82); #endif - /* mcycle: value is derived from the clock with a write offset, and - * freezes on mcountinhibit.CY. Registered after the base callback so - * this one has the last word. */ + /* mcycle/mcycleh: one 64-bit count derived from the clock with a write + * offset, frozen into the register pair while mcountinhibit.CY is set. + * Registered after the base callback so these have the last word. */ this->mcycle.register_callback(std::bind(&Cv32e40pCsr::mcycle_access, this, std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +#if ISS_REG_WIDTH == 32 + this->mcycleh.register_callback(std::bind(&Cv32e40pCsr::mcycleh_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +#endif /* mcountinhibit: reset with all implemented bits set (RTL behaviour), - * i.e. counters disabled out of reset. */ + * i.e. counters disabled out of reset. The callback freezes/unfreezes + * the mcycle count on CY toggles. */ const int num_hpm = CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS; - iss_reg_t mcountinhibit_mask = 0x5 | (((1u << num_hpm) - 1) << 3); - this->mcountinhibit.set_write_mask(mcountinhibit_mask); - this->mcountinhibit.reset_val = mcountinhibit_mask; + this->mcountinhibit.set_write_mask(MCOUNTINHIBIT_MASK); + this->mcountinhibit.reset_val = MCOUNTINHIBIT_MASK; + this->mcountinhibit.register_callback(std::bind(&Cv32e40pCsr::mcountinhibit_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); /* HPM counters and event selectors: only the first num_mhpmcounters * are implemented, the rest are WARL zero (writes ignored). Only @@ -236,25 +242,86 @@ bool Cv32e40pCsr::tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t & return false; } +uint64_t Cv32e40pCsr::mcycle_count() +{ +#if ISS_REG_WIDTH == 32 + uint64_t frozen = ((uint64_t)this->mcycleh.value << 32) | this->mcycle.value; +#else + uint64_t frozen = this->mcycle.value; +#endif + if (this->mcountinhibit.value & 0x1) + { + return frozen; + } + return (uint64_t)((int64_t)this->iss.clock.get_cycles() + this->mcycle_offset); +} + +void Cv32e40pCsr::mcycle_set(uint64_t count) +{ + this->mcycle.value = (iss_reg_t)count; +#if ISS_REG_WIDTH == 32 + this->mcycleh.value = (iss_reg_t)(count >> 32); +#endif + /* While frozen (CY set) this offset is dead: mcountinhibit_access + * recomputes it from the register pair at unfreeze time. */ + this->mcycle_offset = (int64_t)count - (int64_t)this->iss.clock.get_cycles(); +} + bool Cv32e40pCsr::mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) { if (is_write) { - this->mcycle.value = value; - this->mcycle_offset = (int64_t)value - (int64_t)this->iss.clock.get_cycles(); + this->mcycle_set((this->mcycle_count() & ~(uint64_t)0xFFFFFFFF) | value); } else { - if (this->mcountinhibit.value & 0x1) - { - value = this->mcycle.value; - } - else + value = (iss_reg_t)this->mcycle_count(); + } + return false; +} + +bool Cv32e40pCsr::mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->mcycle_set(((uint64_t)value << 32) | (uint32_t)this->mcycle_count()); + } + else + { + value = (iss_reg_t)(this->mcycle_count() >> 32); + } + return false; +} + +bool Cv32e40pCsr::mcountinhibit_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + /* Freeze the count into the register pair when CY sets, re-anchor the + * clock offset to the frozen value when CY clears. Runs before the + * masked store, so this->mcountinhibit.value still holds the old CY. */ + if (is_write) + { + /* Event lines fire only from the full handlers (same scheme as the + * Ri5ky PCMR write): switch when software touches the inhibit CSR. */ + this->iss.exec.switch_to_full_mode(); + bool old_cy = this->mcountinhibit.value & 0x1; + bool new_cy = value & 0x1; + if (old_cy != new_cy) { - value = (iss_reg_t)((int64_t)this->iss.clock.get_cycles() + this->mcycle_offset); + uint64_t count = this->mcycle_count(); + if (new_cy) + { + this->mcycle.value = (iss_reg_t)count; +#if ISS_REG_WIDTH == 32 + this->mcycleh.value = (iss_reg_t)(count >> 32); +#endif + } + else + { + this->mcycle_offset = (int64_t)count - (int64_t)this->iss.clock.get_cycles(); + } } } - return false; + return true; } bool Cv32e40pCsr::hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index) diff --git a/cpu/iss_v2/src/cores/cv32e40p/events.cpp b/cpu/iss_v2/src/cores/cv32e40p/events.cpp new file mode 100644 index 00000000..73a477dc --- /dev/null +++ b/cpu/iss_v2/src/cores/cv32e40p/events.cpp @@ -0,0 +1,16 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#include + +void Cv32e40pEvents::reset(bool active) +{ + Events::reset(active); + if (active) + { + this->pending_events = 0; + } +} diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index 38d05e39..a6f9a93d 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -24,7 +24,7 @@ from typing_extensions import override from gvsoc.systree import Component from cpu.iss_v2.riscv import (RiscvCommon, IssModule, Irq, ExecInOrder, - Regfile, Event, LsuV2, Hwloop) + Regfile, LsuV2, Hwloop) from cpu.iss.isa_gen.isa_gen import Isa, IsaSubset from cpu.iss.isa_gen.isa_riscv_gen import RiscvIsa from cpu.iss.isa_gen.isa_cv32e40pv2 import CoreV2 @@ -42,6 +42,41 @@ class Cv32e40pConfig(RiscvConfig): pass +class Cv32e40pExec(ExecInOrder): + """CV32E40P execution loop: stays on the full handlers while any + implemented counter is enabled, so the event lines fire (same scheme + as Ri5kyExec).""" + + def __init__(self): + super().__init__(scoreboard=True, class_name='Cv32e40pExec', + inorder_commit=True) + + @override + def gen(self, iss: RiscvCommon): + super().gen(iss) + iss.isa.add_include('') + iss.isa.add_implem_include('') + + +class Cv32e40pEvent(IssModule): + """CV32E40P event accounting. + + Selects the Cv32e40pEvents C++ class for the event slot: routes the + architectural event lines (instr, load, store, jump, branch, taken + branch, compressed) into the mhpm counters via Cv32e40pCsr::hpm_commit. + """ + + @override + def gen(self, iss: RiscvCommon): + iss.isa.add_define('CONFIG_GVSOC_ISS_EVENT', 'Cv32e40pEvents') + iss.isa.add_include('') + iss.isa.add_implem_include('') + iss.add_sources([ + 'cpu/iss_v2/src/event/event.cpp', + 'cpu/iss_v2/src/cores/cv32e40p/events.cpp', + ]) + + class Cv32e40pCsr(IssModule): """CV32E40P CSR personality. @@ -131,10 +166,10 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, # standard mcause codes), as implemented by the RTL. The PULP # event-unit style IrqExternal does not apply to this core. 'irq': Irq(), - 'event': Event(), + 'event': Cv32e40pEvent(), 'csr': Cv32e40pCsr(fpu=fpu, zfinx=zfinx, pulp=pulp, num_mhpmcounters=num_mhpmcounters), - 'exec': ExecInOrder(scoreboard=True, inorder_commit=True), + 'exec': Cv32e40pExec(), 'lsu': LsuV2(), 'regfile': Regfile(scoreboard=True), 'hwloop': Hwloop(), From 3a543b3c244b5c35c2fb2146a523d0ef73968261 Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 17 Jul 2026 15:36:51 +0200 Subject: [PATCH 09/28] feat(cv32e40p): co-simulation platform on the iss_v2 core Port of the cv32e40p-standalone co-simulation platform to the io_v2 plane the iss_v2 LSU requires: same memory map (RAM, stdout/timer sinks, debug ROM, virtual exit device, background sparse memory catch-all), rebuilt on router_v2/memory_v3/loader_v2, one target per core configuration (cv32e40p-v2-standalone, -fpu, -zfinx). The exit device and the sparse background memory get io_v2 siblings with the same register map and semantics as the io-plane originals; atomics are rejected explicitly (no A extension on this platform). Memories pass init=False so never-written bytes read 0 like the testbench memory (also fixed on the spike platform, which inherited the poison default). --- cv32e40p-v2-standalone-fpu.py | 37 ++++ cv32e40p-v2-standalone-zfinx.py | 38 ++++ cv32e40p-v2-standalone.py | 37 ++++ .../cv32e40p_exit/cv32e40p_exit_device_v2.cpp | 157 +++++++++++++++ pulp/cv32e40p_exit/cv32e40p_exit_device_v2.py | 35 ++++ .../cv32e40p_sparse_mem_v2.cpp | 94 +++++++++ .../cv32e40p_sparse_mem_v2.py | 36 ++++ pulp/cv32e40p_v2_spike.py | 4 +- pulp/cv32e40p_v2_standalone.py | 188 ++++++++++++++++++ 9 files changed, 625 insertions(+), 1 deletion(-) create mode 100644 cv32e40p-v2-standalone-fpu.py create mode 100644 cv32e40p-v2-standalone-zfinx.py create mode 100644 cv32e40p-v2-standalone.py create mode 100644 pulp/cv32e40p_exit/cv32e40p_exit_device_v2.cpp create mode 100644 pulp/cv32e40p_exit/cv32e40p_exit_device_v2.py create mode 100644 pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.cpp create mode 100644 pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.py create mode 100644 pulp/cv32e40p_v2_standalone.py diff --git a/cv32e40p-v2-standalone-fpu.py b/cv32e40p-v2-standalone-fpu.py new file mode 100644 index 00000000..a2384984 --- /dev/null +++ b/cv32e40p-v2-standalone-fpu.py @@ -0,0 +1,37 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 co-simulation target, FPU configuration (rv32imfc + CoreV). + +import gvsoc.runner +from pulp.cv32e40p_v2_standalone import Cv32e40pStandaloneTop + + +class Cv32e40p(Cv32e40pStandaloneTop): + + def __init__(self, parent, name=None): + super().__init__(parent, name, fpu=1) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 standalone for UVM co-simulation (FPU)" + model = Cv32e40p + name = "cv32e40p-v2-standalone-fpu" diff --git a/cv32e40p-v2-standalone-zfinx.py b/cv32e40p-v2-standalone-zfinx.py new file mode 100644 index 00000000..8fb586c7 --- /dev/null +++ b/cv32e40p-v2-standalone-zfinx.py @@ -0,0 +1,38 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 co-simulation target, ZFINX configuration (rv32imfc + CoreV, +# FP operations on the integer register file). + +import gvsoc.runner +from pulp.cv32e40p_v2_standalone import Cv32e40pStandaloneTop + + +class Cv32e40p(Cv32e40pStandaloneTop): + + def __init__(self, parent, name=None): + super().__init__(parent, name, zfinx=1) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 standalone for UVM co-simulation (ZFINX)" + model = Cv32e40p + name = "cv32e40p-v2-standalone-zfinx" diff --git a/cv32e40p-v2-standalone.py b/cv32e40p-v2-standalone.py new file mode 100644 index 00000000..c7cde8db --- /dev/null +++ b/cv32e40p-v2-standalone.py @@ -0,0 +1,37 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 co-simulation target, base configuration (rv32imc + CoreV). + +import gvsoc.runner +from pulp.cv32e40p_v2_standalone import Cv32e40pStandaloneTop + + +class Cv32e40p(Cv32e40pStandaloneTop): + + def __init__(self, parent, name=None): + super().__init__(parent, name) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 standalone for UVM co-simulation" + model = Cv32e40p + name = "cv32e40p-v2-standalone" diff --git a/pulp/cv32e40p_exit/cv32e40p_exit_device_v2.cpp b/pulp/cv32e40p_exit/cv32e40p_exit_device_v2.cpp new file mode 100644 index 00000000..934ae8af --- /dev/null +++ b/pulp/cv32e40p_exit/cv32e40p_exit_device_v2.cpp @@ -0,0 +1,157 @@ +/* + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +/* + * CV32E40P virtual exit device, io_v2 plane (iss_v2 platforms). + * + * Same register map and semantics as the io-plane sibling + * (cv32e40p_exit_device.cpp): mirrors the UVM virtual peripheral at + * 0x20000000 — status flags (+0x00, magic 123456789 = PASSED, 1 = FAILED), + * exit_valid (+0x04), signature registers (+0x08..+0x10). Writes persist in + * a 256B backing store and read back like testbench memory. + */ + +#include + +#include +#include + +#define VP_STATUS_FLAGS_OFFSET 0x00 +#define VP_EXIT_VALID_OFFSET 0x04 +#define VP_SIG_START_OFFSET 0x08 +#define VP_SIG_END_OFFSET 0x0C +#define VP_SIG_WRITE_OFFSET 0x10 + +class Cv32e40pExitDeviceV2 : public vp::Component +{ +public: + Cv32e40pExitDeviceV2(vp::ComponentConf &config); + +private: + static vp::IoReqStatus req(vp::Block *__this, vp::IoReq *req); + + vp::Trace trace; + vp::IoSlave in{&Cv32e40pExitDeviceV2::req}; + + /* Backing store (256B region): writes persist and read back, like the + * testbench memory under the virtual peripheral. */ + uint8_t mem[0x100] = {}; +}; + +Cv32e40pExitDeviceV2::Cv32e40pExitDeviceV2(vp::ComponentConf &config) + : vp::Component(config) +{ + this->traces.new_trace("trace", &this->trace, vp::DEBUG); + this->new_slave_port("input", &this->in); +} + +vp::IoReqStatus Cv32e40pExitDeviceV2::req(vp::Block *__this, vp::IoReq *req) +{ + Cv32e40pExitDeviceV2 *_this = (Cv32e40pExitDeviceV2 *)__this; + + uint32_t offset = (uint32_t)req->get_addr(); + + /* Atomics are not supported (no A extension on this platform). */ + if (req->get_opcode() != vp::IoReqOpcode::READ && + req->get_opcode() != vp::IoReqOpcode::WRITE) + { + req->set_resp_status(vp::IO_RESP_INVALID); + return vp::IO_REQ_DONE; + } + + if (req->get_opcode() == vp::IoReqOpcode::READ) + { + if (offset + req->get_size() <= sizeof(_this->mem)) + memcpy(req->get_data(), &_this->mem[offset], req->get_size()); + else + memset(req->get_data(), 0, req->get_size()); + return vp::IO_REQ_DONE; + } + + if (offset + req->get_size() <= sizeof(_this->mem)) + memcpy(&_this->mem[offset], req->get_data(), req->get_size()); + + uint32_t wdata = (req->get_size() == 4) ? *(uint32_t *)req->get_data() : 0; + + switch (offset) + { + case VP_STATUS_FLAGS_OFFSET: + if (wdata == 123456789U) /* 0x075BCD15 — TEST PASSED (same magic as UVM VP) */ + { + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "vp_status write: wdata=0x%08x — TEST PASSED, stopping simulation\n", wdata); + fprintf(stdout, "[cv32e40p_exit] tests_passed=1 exit_value=0x00000000\n"); + fflush(stdout); + _this->time.get_engine()->quit(0); + } + else if (wdata == 1U) /* TEST FAILED */ + { + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "vp_status write: wdata=0x%08x — TEST FAILED, stopping simulation\n", wdata); + fprintf(stdout, "[cv32e40p_exit] tests_failed=1 exit_value=0x00000001\n"); + fflush(stdout); + _this->time.get_engine()->quit(1); + } + else + { + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "vp_status write: wdata=0x%08x (unrecognized status flag — ignored)\n", wdata); + } + break; + + case VP_EXIT_VALID_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "exit_valid asserted: exit_value=0x%08x — stopping simulation\n", wdata); + fprintf(stdout, "[cv32e40p_exit] exit_valid=1 exit_value=0x%08x\n", wdata); + fflush(stdout); + _this->time.get_engine()->quit((int32_t)wdata); + break; + + case VP_SIG_START_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "signature_start_address=0x%08x (not implemented)\n", wdata); + break; + + case VP_SIG_END_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "signature_end_address=0x%08x (not implemented)\n", wdata); + break; + + case VP_SIG_WRITE_OFFSET: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "signature write triggered — stopping simulation (exit_value=0)\n"); + fprintf(stdout, "[cv32e40p_exit] signature write → exit_valid=1 exit_value=0x00000000\n"); + fflush(stdout); + _this->time.get_engine()->quit(0); + break; + + default: + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "unknown offset 0x%02x wdata=0x%08x — ignored\n", offset, wdata); + break; + } + + return vp::IO_REQ_DONE; +} + +extern "C" vp::Component *gv_new(vp::ComponentConf &config) +{ + return new Cv32e40pExitDeviceV2(config); +} diff --git a/pulp/cv32e40p_exit/cv32e40p_exit_device_v2.py b/pulp/cv32e40p_exit/cv32e40p_exit_device_v2.py new file mode 100644 index 00000000..3e23a090 --- /dev/null +++ b/pulp/cv32e40p_exit/cv32e40p_exit_device_v2.py @@ -0,0 +1,35 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P virtual exit device on the io_v2 plane: same register map as the +# io-plane sibling (UVM virtual peripheral at 0x20000000). + +import gvsoc.systree +import gvsoc.signature + + +class Cv32e40pExitDeviceV2(gvsoc.systree.Component): + + def __init__(self, parent, name): + super().__init__(parent, name) + self.add_sources(['pulp/cv32e40p_exit/cv32e40p_exit_device_v2.cpp']) + + def i_INPUT(self) -> gvsoc.systree.SlaveItf: + return gvsoc.systree.SlaveItf(self, 'input', signature=gvsoc.signature.IoV2Sync()) diff --git a/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.cpp b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.cpp new file mode 100644 index 00000000..21f48129 --- /dev/null +++ b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.cpp @@ -0,0 +1,94 @@ +/* + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +/* + * Background sparse memory, io_v2 plane (iss_v2 platforms). + * + * Same contract as the io-plane sibling (cv32e40p_sparse_mem.cpp): mapped as + * the interconnect's catch-all route, it makes never-written bytes read 0 and + * writes persist, matching the UVM testbench sparse memory model. + */ + +#include +#include + +#include +#include + +class Cv32e40pSparseMemV2 : public vp::Component +{ +public: + Cv32e40pSparseMemV2(vp::ComponentConf &config); + +private: + static vp::IoReqStatus req(vp::Block *__this, vp::IoReq *req); + + vp::Trace trace; + vp::IoSlave in{&Cv32e40pSparseMemV2::req}; + + /* Byte-granular backing store: only written bytes are kept. */ + std::unordered_map store; +}; + +Cv32e40pSparseMemV2::Cv32e40pSparseMemV2(vp::ComponentConf &config) + : vp::Component(config) +{ + this->traces.new_trace("trace", &this->trace, vp::DEBUG); + this->new_slave_port("input", &this->in); +} + +vp::IoReqStatus Cv32e40pSparseMemV2::req(vp::Block *__this, vp::IoReq *req) +{ + Cv32e40pSparseMemV2 *_this = (Cv32e40pSparseMemV2 *)__this; + + uint64_t addr = req->get_addr(); + uint64_t size = req->get_size(); + uint8_t *data = req->get_data(); + + _this->trace.msg(vp::Trace::LEVEL_DEBUG, + "background access (addr: 0x%llx, size: 0x%llx, is_write: %d)\n", + addr, size, req->get_is_write()); + + if (req->get_opcode() == vp::IoReqOpcode::WRITE) + { + for (uint64_t i = 0; i < size; i++) + _this->store[addr + i] = data[i]; + } + else if (req->get_opcode() == vp::IoReqOpcode::READ) + { + for (uint64_t i = 0; i < size; i++) + { + auto it = _this->store.find(addr + i); + data[i] = (it != _this->store.end()) ? it->second : 0; + } + } + else + { + /* Atomics are not supported (no A extension on this platform). */ + req->set_resp_status(vp::IO_RESP_INVALID); + } + + return vp::IO_REQ_DONE; +} + +extern "C" vp::Component *gv_new(vp::ComponentConf &config) +{ + return new Cv32e40pSparseMemV2(config); +} diff --git a/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.py b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.py new file mode 100644 index 00000000..1e15e239 --- /dev/null +++ b/pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.py @@ -0,0 +1,36 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# Background sparse memory on the io_v2 plane: catch-all route for iss_v2 +# platforms, same contract as the io-plane sibling (never-written bytes read +# 0, writes persist, like the UVM testbench sparse memory model). + +import gvsoc.systree +import gvsoc.signature + + +class Cv32e40pSparseMemV2(gvsoc.systree.Component): + + def __init__(self, parent, name): + super().__init__(parent, name) + self.add_sources(['pulp/cv32e40p_sparse_mem/cv32e40p_sparse_mem_v2.cpp']) + + def i_INPUT(self) -> gvsoc.systree.SlaveItf: + return gvsoc.systree.SlaveItf(self, 'input', signature=gvsoc.signature.IoV2Sync()) diff --git a/pulp/cv32e40p_v2_spike.py b/pulp/cv32e40p_v2_spike.py index 7dd72e0b..b8c79e8c 100644 --- a/pulp/cv32e40p_v2_spike.py +++ b/pulp/cv32e40p_v2_spike.py @@ -102,7 +102,9 @@ def __post_init__(self): # recipe to match the RTL compressed decoder. isa = 'rv32imfc' if (self.fpu or self.zfinx) else 'rv32imc' self.core = Cv32e40pConfig(isa=isa, boot_addr=self.boot_addr) - self.mem = MemoryV3Config('mem', size=self.mem_size, atomics=False, latency=1) + # init=False: unwritten memory reads 0, not the 0x57 poison default. + self.mem = MemoryV3Config('mem', size=self.mem_size, atomics=False, latency=1, + init=False) self.router = RouterConfig(kind='bandwidth') self.mem_mapping = RouterMapping(name='mem_mapping', base=self.mem_base, size=self.mem_size) diff --git a/pulp/cv32e40p_v2_standalone.py b/pulp/cv32e40p_v2_standalone.py new file mode 100644 index 00000000..072f1147 --- /dev/null +++ b/pulp/cv32e40p_v2_standalone.py @@ -0,0 +1,188 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 standalone platform for UVM co-simulation, shared by the +# cv32e40p-v2-standalone* targets (one target name per core configuration). +# +# Same memory map as the io-plane co-sim platform (cv32e40p-standalone.py), +# rebuilt on the io_v2 plane the iss_v2 LSU requires: +# 0x00000000 4MB Main RAM (entry point 0x00000080) +# 0x10000000 256B Virtual STDOUT (write-only sink) +# 0x15000000 256B Virtual TIMER (write-only sink) +# 0x1A110800 4KB Debug ROM (linker script `dbg` region) +# 0x20000000 256B Virtual EXIT (terminates the simulation) +# everywhere else background sparse memory (catch-all route: reads 0 +# until written, writes persist - same as the testbench) + +import vp.clock_domain +import gvsoc.systree +from gvrun.parameter import TargetParameter +from config_tree import Config, cfg_field +from memory.memory_v3 import Memory, MemoryV3Config +from interco.router_v2 import Router, RouterConfig, RouterMapping +from utils.loader.loader_v2 import ElfLoader +from pulp.cv32e40p_exit.cv32e40p_exit_device_v2 import Cv32e40pExitDeviceV2 +from pulp.cv32e40p_sparse_mem.cv32e40p_sparse_mem_v2 import Cv32e40pSparseMemV2 +from pulp.cpu.iss.cv32e40p_v2 import Cv32e40p as Cv32e40pV2Core, Cv32e40pConfig + + +class Cv32e40pStandaloneConfig(Config): + """Configuration for the CV32E40P iss_v2 co-simulation SoC.""" + + boot_addr: int = cfg_field(default=0x80, fmt="hex", dump=True, desc=( + "Boot address (matches RTL BOOT_ADDR)" + )) + + fpu: int = cfg_field(default=0, dump=True, desc=( + "FPU configuration (F extension, FP register file)" + )) + + zfinx: int = cfg_field(default=0, dump=True, desc=( + "ZFINX configuration (FP operations on the integer register file)" + )) + + core: Cv32e40pConfig = cfg_field(init=False, desc=( + "CV32E40P core configuration" + )) + + mem: MemoryV3Config = cfg_field(init=False, desc=( + "Main RAM configuration" + )) + + stdout: MemoryV3Config = cfg_field(init=False, desc=( + "Virtual STDOUT sink configuration" + )) + + timer: MemoryV3Config = cfg_field(init=False, desc=( + "Virtual TIMER sink configuration" + )) + + debug_rom: MemoryV3Config = cfg_field(init=False, desc=( + "Debug ROM configuration" + )) + + router: RouterConfig = cfg_field(init=False, desc=( + "Router configuration" + )) + + mem_mapping: RouterMapping = cfg_field(init=False, desc=( + "Main RAM address range" + )) + + stdout_mapping: RouterMapping = cfg_field(init=False, desc=( + "Virtual STDOUT address range" + )) + + timer_mapping: RouterMapping = cfg_field(init=False, desc=( + "Virtual TIMER address range" + )) + + debug_rom_mapping: RouterMapping = cfg_field(init=False, desc=( + "Debug ROM address range" + )) + + exit_mapping: RouterMapping = cfg_field(init=False, desc=( + "Virtual EXIT device address range" + )) + + background_mapping: RouterMapping = cfg_field(init=False, desc=( + "Background sparse memory catch-all route" + )) + + def __post_init__(self): + super().__post_init__() + # ZFINX needs the F opcodes in the decoder (routed to the integer + # register file); compressed FP rows are disabled by the core recipe. + isa = 'rv32imfc' if (self.fpu or self.zfinx) else 'rv32imc' + self.core = Cv32e40pConfig(isa=isa, boot_addr=self.boot_addr) + # init=False: never-written bytes must read 0 (testbench memory + # contract), not the 0x57 poison pattern of the default init=True. + self.mem = MemoryV3Config('mem', size=0x0040_0000, atomics=False, latency=1, + init=False) + self.stdout = MemoryV3Config('stdout', size=0x100, atomics=False, latency=1, + init=False) + self.timer = MemoryV3Config('timer', size=0x100, atomics=False, latency=1, + init=False) + self.debug_rom = MemoryV3Config('debug_rom', size=0x1000, atomics=False, latency=1, + init=False) + self.router = RouterConfig(kind='bandwidth') + self.mem_mapping = RouterMapping(name='mem_mapping', + base=0x0000_0000, size=0x0040_0000) + self.stdout_mapping = RouterMapping(name='stdout_mapping', + base=0x1000_0000, size=0x100) + self.timer_mapping = RouterMapping(name='timer_mapping', + base=0x1500_0000, size=0x100) + self.debug_rom_mapping = RouterMapping(name='debug_rom_mapping', + base=0x1A11_0800, size=0x1000) + self.exit_mapping = RouterMapping(name='exit_mapping', + base=0x2000_0000, size=0x100) + # Catch-all (size=0): absolute addresses forwarded so the sparse + # store is indexed like the testbench's. + self.background_mapping = RouterMapping(name='background_mapping', + base=0x0000_0000, size=0, + remove_base=False) + + +class Cv32e40pStandaloneSoc(gvsoc.systree.Component): + + def __init__(self, parent, name, config: Cv32e40pStandaloneConfig, binary): + super().__init__(parent, name, config=config) + + mem = Memory ( self, 'mem' , config=config.mem ) + stdout = Memory ( self, 'stdout' , config=config.stdout ) + timer = Memory ( self, 'timer' , config=config.timer ) + dbgrom = Memory ( self, 'debug_rom', config=config.debug_rom ) + exit_d = Cv32e40pExitDeviceV2( self, 'exit' ) + bg_mem = Cv32e40pSparseMemV2 ( self, 'background_mem' ) + ico = Router ( self, 'ico' , config=config.router ) + core = Cv32e40pV2Core ( self, 'core' , config=config.core , + fpu=bool(config.fpu), zfinx=bool(config.zfinx) ) + loader = ElfLoader ( self, 'loader' , binary=binary ) + + ico.o_MAP ( mem.i_INPUT() , mapping=config.mem_mapping ) + ico.o_MAP ( stdout.i_INPUT(), mapping=config.stdout_mapping ) + ico.o_MAP ( timer.i_INPUT() , mapping=config.timer_mapping ) + ico.o_MAP ( dbgrom.i_INPUT(), mapping=config.debug_rom_mapping ) + ico.o_MAP ( exit_d.i_INPUT(), mapping=config.exit_mapping ) + ico.o_MAP ( bg_mem.i_INPUT(), mapping=config.background_mapping ) + + # Three independent masters, one router input port each. + loader.o_OUT ( ico.i_INPUT(0) ) + loader.o_START ( core.i_FETCHEN() ) + loader.o_ENTRY ( core.i_ENTRY() ) + + core.o_FETCH ( ico.i_INPUT(1) ) + core.o_DATA ( ico.i_INPUT(2) ) + + +class Cv32e40pStandaloneTop(gvsoc.systree.Component): + + def __init__(self, parent, name=None, fpu: int=0, zfinx: int=0): + super().__init__(parent, name) + + binary = TargetParameter( + self, name='binary', value=None, description='ELF binary to simulate' + ).get_value() + + config = Cv32e40pStandaloneConfig('soc', fpu=fpu, zfinx=zfinx) + + clock = vp.clock_domain.Clock_domain(self, 'clock', frequency=50000000) + soc = Cv32e40pStandaloneSoc(self, 'soc', config, binary) + clock.o_CLOCK(soc.i_CLOCK()) From ef012e7b391035d5e4363dad577d56c7540d005f Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 17 Jul 2026 17:22:42 +0200 Subject: [PATCH 10/28] feat(cv32e40p): architectural commit stream and mtvec WARL fixup on the v2 core An external stepper (RVVI bridge) samples the register file at each retire, but the retire hook fires at issue while an async load writes its rd only with the LSU response: expose a commit stream instead - a PC ring pushed once the writeback is architecturally visible, at commit-FIFO drain for held instructions - and pin the core to the full dispatch path while the stream is observed (the fast path skips the FIFO bookkeeping the stream is built on). Restore the RTL WARL result of mtvec over the generic IrqRiscv callback, which drops the mode bit the RTL keeps (cv32e40p_cs_registers.sv:667). Boot the co-simulation platform at the fixed BOOT_ADDR like the RTL: the ELF-entry sync would also rewrite mtvec after a CSR injection. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 1 + cpu/iss_v2/include/cores/cv32e40p/events.hpp | 22 ++++++++++++ .../include/cores/cv32e40p/events_implem.hpp | 35 +++++++++++++++++++ cpu/iss_v2/include/cores/cv32e40p/exec.hpp | 5 +++ .../include/cores/cv32e40p/exec_implem.hpp | 2 ++ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 23 ++++++++++++ cpu/iss_v2/src/cores/cv32e40p/events.cpp | 4 +++ pulp/cv32e40p_v2_standalone.py | 4 ++- 8 files changed, 95 insertions(+), 1 deletion(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index b4eacb89..b2242417 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -128,6 +128,7 @@ class Cv32e40pCsr : public Csr bool mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcountinhibit_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mstatus_read_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool mtvec_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); /* Current 64-bit mcycle count: the frozen register pair while * mcountinhibit.CY is set, the offset clock otherwise. */ diff --git a/cpu/iss_v2/include/cores/cv32e40p/events.hpp b/cpu/iss_v2/include/cores/cv32e40p/events.hpp index 863dab03..9a1ed18e 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events.hpp @@ -33,8 +33,30 @@ class Cv32e40pEvents : public Events inline void event_taken_branch_account(); inline void event_jump_account(); inline void event_retire_account(iss_insn_t *insn); + inline void insn_stall_account(); + + /* Architectural commit stream for an external stepper (RVVI bridge). + * A PC is pushed when the instruction's result is architecturally + * visible: at retire for sync instructions, at commit-FIFO drain for + * held ones (async load, WFI) - insn_stall_account only fires there - + * where the writeback has already happened. The stepper pops. Sampling + * the regfile on the raw retire hook instead would race the load + * writeback (the LSU response lands one cycle later). */ + static constexpr int COMMIT_RING = 64; + uint64_t commit_push = 0; + uint64_t commit_pop = 0; + iss_reg_t commit_pc[COMMIT_RING]; + + /* Drop commits not consumed yet (external resync forced a new PC). */ + inline void commit_stream_flush(); private: + /* Program-order PCs of the instructions parked in the exec commit + * FIFO (held or sync follower); drain pops them in the same order. */ + uint64_t inflight_push = 0; + uint64_t inflight_pop = 0; + iss_reg_t inflight_pc[COMMIT_RING]; + /* Event lines fired by the executing instruction, committed as one OR * mask at retire: each counter advances at most +1 per instruction, as * the RTL advances at most +1 per cycle (cv32e40p_cs_registers.sv:1437). diff --git a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp index 751b4b8c..47d9b179 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp @@ -46,6 +46,21 @@ inline void Cv32e40pEvents::event_jump_account() inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) { Events::event_retire_account(insn); +#ifdef CONFIG_GVSOC_ISS_EXEC_INORDER_COMMIT + if (this->iss.exec.queue_head != NULL) + { + /* Parked in the commit FIFO (held, or sync follower behind a held + * head): visible at drain time, through insn_stall_account, in + * this same program order. */ + this->inflight_pc[this->inflight_push % COMMIT_RING] = insn->addr; + this->inflight_push++; + } + else +#endif + { + this->commit_pc[this->commit_push % COMMIT_RING] = insn->addr; + this->commit_push++; + } /* A trapping instruction does not retire: drop its event lines. */ if (this->iss.exec.has_exception) { @@ -57,3 +72,23 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) this->pending_events = 0; this->iss.csr.hpm_commit(events); } + +inline void Cv32e40pEvents::insn_stall_account() +{ + /* Fires only from ExecInOrder::drain_entry: one commit-FIFO entry + * retired, writeback done. The guard covers entries flushed by + * commit_stream_flush while their drain was still pending. */ + if (this->inflight_pop < this->inflight_push) + { + this->commit_pc[this->commit_push % COMMIT_RING] = + this->inflight_pc[this->inflight_pop % COMMIT_RING]; + this->inflight_pop++; + this->commit_push++; + } +} + +inline void Cv32e40pEvents::commit_stream_flush() +{ + this->commit_pop = this->commit_push; + this->inflight_pop = this->inflight_push; +} diff --git a/cpu/iss_v2/include/cores/cv32e40p/exec.hpp b/cpu/iss_v2/include/cores/cv32e40p/exec.hpp index e76c5bb3..2e94bbaf 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/exec.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/exec.hpp @@ -16,4 +16,9 @@ class Cv32e40pExec : public ExecInOrder Cv32e40pExec(Iss &iss) : ExecInOrder(iss) {} inline bool can_switch_to_fast_mode(); + + /* Set by an external observer (RVVI bridge) consuming the commit + * stream: the fast dispatch path skips the commit-FIFO bookkeeping + * the stream is built on, so stay on the full handlers. */ + bool commit_stream_observed = false; }; diff --git a/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp index 288d6809..a6bcfa95 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/exec_implem.hpp @@ -13,6 +13,8 @@ inline bool Cv32e40pExec::can_switch_to_fast_mode() { if (!ExecInOrder::can_switch_to_fast_mode()) return false; + if (this->commit_stream_observed) return false; + /* The event lines only fire from the full handlers: stay there while * any implemented counter is enabled (see Cv32e40pCsr::hpm_counting). */ return !this->iss.csr.hpm_counting(); diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index d70fe499..664c4c52 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -201,6 +201,12 @@ void Cv32e40pCsr::start() * overwrites the read value with the stored one. */ this->mstatus.register_callback(std::bind(&Cv32e40pCsr::mstatus_read_fixup, this, std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + + /* Registered here so it runs after IrqRiscv::mtvec_access, which is + * registered by the IrqRiscv constructor and stores the value with the + * mode bit cleared. */ + this->mtvec.register_callback(std::bind(&Cv32e40pCsr::mtvec_write_fixup, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); } void Cv32e40pCsr::reset(bool active) @@ -232,6 +238,23 @@ bool Cv32e40pCsr::mstatus_read_fixup(iss_insn_t *insn, bool is_write, iss_reg_t return false; } +bool Cv32e40pCsr::mtvec_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (!is_write) + { + /* Keep the default read (value = stored register). */ + return true; + } + + /* RTL WARL result (cv32e40p_cs_registers.sv): base = wdata[31:8], + * bits [7:1] read 0, mode = wdata[0]; the reset / boot-address path + * (insn == NULL) forces mode = 1 (MTVEC_MODE). IrqRiscv::mtvec_access + * ran before this and stored the value with the mode bit cleared. */ + iss_reg_t mode = (insn == NULL) ? 1 : (value & 1); + this->mtvec.value = (value & 0xFFFFFF00) | mode; + return false; +} + bool Cv32e40pCsr::tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value) { /* One trigger: tselect always reads 0 (the base callback reads -1). */ diff --git a/cpu/iss_v2/src/cores/cv32e40p/events.cpp b/cpu/iss_v2/src/cores/cv32e40p/events.cpp index 73a477dc..1b418aaa 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/events.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/events.cpp @@ -12,5 +12,9 @@ void Cv32e40pEvents::reset(bool active) if (active) { this->pending_events = 0; + this->commit_push = 0; + this->commit_pop = 0; + this->inflight_push = 0; + this->inflight_pop = 0; } } diff --git a/pulp/cv32e40p_v2_standalone.py b/pulp/cv32e40p_v2_standalone.py index 072f1147..75da31ba 100644 --- a/pulp/cv32e40p_v2_standalone.py +++ b/pulp/cv32e40p_v2_standalone.py @@ -164,9 +164,11 @@ def __init__(self, parent, name, config: Cv32e40pStandaloneConfig, binary): ico.o_MAP ( bg_mem.i_INPUT(), mapping=config.background_mapping ) # Three independent masters, one router input port each. + # o_ENTRY is deliberately NOT bound: like the RTL, the core boots at + # the fixed BOOT_ADDR (config boot_addr), not at the ELF entry; the + # entry sync would also rewrite mtvec after a co-sim CSR injection. loader.o_OUT ( ico.i_INPUT(0) ) loader.o_START ( core.i_FETCHEN() ) - loader.o_ENTRY ( core.i_ENTRY() ) core.o_FETCH ( ico.i_INPUT(1) ) core.o_DATA ( ico.i_INPUT(2) ) From fe25febf55081e8839852c2622bdc1f8bc768c1f Mon Sep 17 00:00:00 2001 From: mpaci Date: Sat, 18 Jul 2026 14:19:56 +0200 Subject: [PATCH 11/28] feat(cv32e40p): IRQ wire delivery, trap/priv personalities and RTL decode fixes on the v2 core - Cv32e40pIrq personality: RTL priority ladder (31..16, MEI, MSI, MTI), vectored entry from mtvec{base,mode}, single-source IRQ_MASK and a mie write fixup (the generic mie_access bypasses the write mask). - cv32e40p_irq_injector component: 19 wire masters driven through external_bind; the standalone platform binds them to core.i_IRQ(n). - Cv32e40pCore / Cv32e40pException personalities: mcause stays sticky across mret and exception entry masks the mtvec mode bits. - Cv32e40pRegfile: one-shot write-back suppression armed by the FP trap sites (CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS, set by the recipe on fpu/zfinx variants). - Csr personality: user counter aliases (cycle/instret/hpm and their high views), mip front-end view (reads mirror the wire-driven store, writes dropped as on the RTL), architectural hwloop LPEND shadow, mip cleared in reset, FS not dirtied by trapped FP instructions. - Events: trap sequence stamps on redirected commits so the bridge can gate the deferred state compare across a drain window. - priv.hpp: CV32E40P csr dispatch where csrrc/csrrs with rs1=x0 and csrrsi/csrrci with uimm=0 are reads, not writes. - Recipe: fence/fence.i decode relaxed to funct3-only as in the RTL decoder (reserved fields ignored), pulp parameter and the cv32e40p-v2-standalone-nopulp target (rv32imc, no CoreV, no X bit). --- cpu/iss_v2/include/cores/cv32e40p/core.hpp | 23 ++ cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 53 +++- cpu/iss_v2/include/cores/cv32e40p/events.hpp | 14 +- .../include/cores/cv32e40p/events_implem.hpp | 4 + .../include/cores/cv32e40p/exception.hpp | 23 ++ cpu/iss_v2/include/cores/cv32e40p/irq.hpp | 30 +++ cpu/iss_v2/include/cores/cv32e40p/priv.hpp | 228 ++++++++++++++++++ cpu/iss_v2/include/cores/cv32e40p/regfile.hpp | 56 +++++ cpu/iss_v2/src/cores/cv32e40p/core.cpp | 17 ++ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 135 ++++++++++- cpu/iss_v2/src/cores/cv32e40p/exception.cpp | 24 ++ cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 104 ++++++++ cv32e40p-v2-standalone-nopulp.py | 37 +++ pulp/cpu/iss/cv32e40p_v2.py | 117 ++++++++- .../cv32e40p_irq_injector.cpp | 108 +++++++++ .../cv32e40p_irq_injector.py | 42 ++++ pulp/cv32e40p_v2_standalone.py | 20 +- 17 files changed, 1010 insertions(+), 25 deletions(-) create mode 100644 cpu/iss_v2/include/cores/cv32e40p/core.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/exception.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/irq.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/priv.hpp create mode 100644 cpu/iss_v2/include/cores/cv32e40p/regfile.hpp create mode 100644 cpu/iss_v2/src/cores/cv32e40p/core.cpp create mode 100644 cpu/iss_v2/src/cores/cv32e40p/exception.cpp create mode 100644 cpu/iss_v2/src/cores/cv32e40p/irq.cpp create mode 100644 cv32e40p-v2-standalone-nopulp.py create mode 100644 pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp create mode 100644 pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.py diff --git a/cpu/iss_v2/include/cores/cv32e40p/core.hpp b/cpu/iss_v2/include/cores/cv32e40p/core.hpp new file mode 100644 index 00000000..0c5b26d8 --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/core.hpp @@ -0,0 +1,23 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include +#include + +class Cv32e40pCore : public Core +{ +public: + Cv32e40pCore(Iss &iss) : Core(iss), iss(iss) {} + + /* MRET with the mcause hold-over of the RTL; shadows the generic + * handler (static dispatch via CONFIG_GVSOC_ISS_CORE). */ + iss_reg_t mret_handle(); + +private: + Iss &iss; +}; diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index b2242417..725ac23b 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -61,6 +61,16 @@ class Cv32e40pHwloopCsr : public CsrAbtractReg bool check_access(Iss *iss, bool write, bool read) override; }; +/* User-mode counter alias (cycle/instret/hpmcounterN and the H views): + * reads mirror the machine counter through a registered callback, writes + * raise illegal-instruction (0xCxx is the architecturally read-only CSR + * range, and the RTL has no write path for it). */ +class Cv32e40pCounterAlias : public CsrAbtractReg +{ +public: + bool check_access(Iss *iss, bool write, bool read) override; +}; + class Cv32e40pCsr : public Csr { public: @@ -78,8 +88,9 @@ class Cv32e40pCsr : public Csr /* Promote mstatus.FS to Dirty (11) on FP state change. The RTL forces it * on FP regfile writes, fflags updates and FP-CSR writes when the FPU is - * in the ISA (FPU=1, ZFINX=0). SD (bit 31) is derived at read time. */ - inline void fp_state_dirty(); + * in the ISA (FPU=1, ZFINX=0). SD (bit 31) is derived at read time. + * Out-of-line: the trapped-instruction guard needs the full Iss type. */ + void fp_state_dirty(); /* Advance the counters for one retired instruction: events is the OR of * the RTL hpm_events lines it fired (see cores/cv32e40p/events.hpp). @@ -113,6 +124,30 @@ class Cv32e40pCsr : public Csr /* Hardware-loop CSRs: 0xCC0..0xCC2 / 0xCC4..0xCC6 (gap at 0xCC3). */ Cv32e40pHwloopCsr hwloop_csr[6]; + /* Architectural LPEND per loop, written by the corev.hpp setters. The + * Hwloop module stores the loop-back point (LPEND - 4), so it cannot + * serve the CSR read: a never-programmed loop must read back 0. */ + iss_reg_t hwloop_lpend[2] = {0, 0}; + + /* mip front-end (0x344): reads mirror the wire-driven base register, + * CSR writes are silently dropped — the RTL has no mip write path + * (cv32e40p_cs_registers.sv reads it from the interrupt lines only). + * Replaces the base mip in the CSR map, so the generic IrqRiscv write + * callback (wdata & 0xAAA, which also clears the fast-line bits) can + * never corrupt the pending state. */ + CsrAbtractReg mip_view; + + /* User counter aliases: 0xC00/0xC02/0xC03..0xC1F and the H views at + * 0xC80/0xC82/0xC83..0xC9F. time (0xC01) is absent. */ + Cv32e40pCounterAlias cycle_alias; + Cv32e40pCounterAlias instret_alias; + Cv32e40pCounterAlias hpmcounter_alias[29]; +#if ISS_REG_WIDTH == 32 + Cv32e40pCounterAlias cycleh_alias; + Cv32e40pCounterAlias instreth_alias; + Cv32e40pCounterAlias hpmcounterh_alias[29]; +#endif + /* fflags / frm / fcsr (0x001..0x003). */ Cv32e40pFpCsr fflags_csr; Cv32e40pFpCsr frm_csr; @@ -124,8 +159,15 @@ class Cv32e40pCsr : public Csr bool fcsr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); bool tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool cycle_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool cycleh_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool instret_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool instreth_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool hpm_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); + bool hpmh_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); bool mcountinhibit_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mstatus_read_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mtvec_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); @@ -152,13 +194,6 @@ inline bool Cv32e40pCsr::fp_access_illegal() #endif } -inline void Cv32e40pCsr::fp_state_dirty() -{ -#if CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA - this->mstatus.fs = 3; -#endif -} - inline bool Cv32e40pCsr::hpm_counting() { /* CY excluded: mcycle is clock-derived and needs no full-handler diff --git a/cpu/iss_v2/include/cores/cv32e40p/events.hpp b/cpu/iss_v2/include/cores/cv32e40p/events.hpp index 9a1ed18e..34489a97 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events.hpp @@ -47,15 +47,27 @@ class Cv32e40pEvents : public Events uint64_t commit_pop = 0; iss_reg_t commit_pc[COMMIT_RING]; + /* Trap-redirect sequence, bumped by Cv32e40pException::raise and the + * Cv32e40pIrq take. Each commit entry is stamped with the value seen + * when the instruction executed; a stamp older than the current + * trap_seq tells the stepper the sampled CSR/GPR state already + * includes a later redirect (a trap taken while this entry was still + * held in the commit FIFO), so state compares on it must be skipped. */ + uint64_t trap_seq = 0; + uint64_t commit_trap_seq[COMMIT_RING]; + /* Drop commits not consumed yet (external resync forced a new PC). */ inline void commit_stream_flush(); private: /* Program-order PCs of the instructions parked in the exec commit - * FIFO (held or sync follower); drain pops them in the same order. */ + * FIFO (held or sync follower); drain pops them in the same order. + * The trap_seq stamp is taken here, at execution, and carried to the + * commit entry at drain. */ uint64_t inflight_push = 0; uint64_t inflight_pop = 0; iss_reg_t inflight_pc[COMMIT_RING]; + uint64_t inflight_trap_seq[COMMIT_RING]; /* Event lines fired by the executing instruction, committed as one OR * mask at retire: each counter advances at most +1 per instruction, as diff --git a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp index 47d9b179..64fde14b 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp @@ -53,12 +53,14 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) * head): visible at drain time, through insn_stall_account, in * this same program order. */ this->inflight_pc[this->inflight_push % COMMIT_RING] = insn->addr; + this->inflight_trap_seq[this->inflight_push % COMMIT_RING] = this->trap_seq; this->inflight_push++; } else #endif { this->commit_pc[this->commit_push % COMMIT_RING] = insn->addr; + this->commit_trap_seq[this->commit_push % COMMIT_RING] = this->trap_seq; this->commit_push++; } /* A trapping instruction does not retire: drop its event lines. */ @@ -82,6 +84,8 @@ inline void Cv32e40pEvents::insn_stall_account() { this->commit_pc[this->commit_push % COMMIT_RING] = this->inflight_pc[this->inflight_pop % COMMIT_RING]; + this->commit_trap_seq[this->commit_push % COMMIT_RING] = + this->inflight_trap_seq[this->inflight_pop % COMMIT_RING]; this->inflight_pop++; this->commit_push++; } diff --git a/cpu/iss_v2/include/cores/cv32e40p/exception.hpp b/cpu/iss_v2/include/cores/cv32e40p/exception.hpp new file mode 100644 index 00000000..8def012e --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/exception.hpp @@ -0,0 +1,23 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include +#include + +class Cv32e40pException : public Exception +{ +public: + Cv32e40pException(Iss &iss) : Exception(iss), iss(iss) {} + + /* Exception entry at the mtvec base; shadows the generic raise + * (static dispatch via CONFIG_GVSOC_ISS_EXCEPTION). */ + void raise(iss_reg_t pc, int id); + +private: + Iss &iss; +}; diff --git a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp new file mode 100644 index 00000000..f1477643 --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp @@ -0,0 +1,30 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include +#include + +class Cv32e40pIrq : public IrqRiscv +{ +public: + /* Interrupt lines wired in the RTL: MSI(3), MTI(7), MEI(11) and the + * sixteen fast lines irq[31:16] (cv32e40p_int_controller.sv IRQ_MASK). */ + static constexpr iss_reg_t IRQ_MASK = 0xFFFF0888; + + Cv32e40pIrq(Iss &iss) : IrqRiscv(iss) {} + + void start(); + + /* Interrupt take with the RTL priority order and vectored entry; + * shadows the generic RISC-V ladder (static dispatch via + * CONFIG_GVSOC_ISS_IRQ). */ + int check(); + +private: + bool mie_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); +}; diff --git a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp new file mode 100644 index 00000000..20f754a1 --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp @@ -0,0 +1,228 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +/* CV32E40P privileged-instruction handlers, replacing + * for the whole priv subset (the recipe + * swaps the subset include, so this file provides the full surface: + * csrr*, wfi, xret, sfence.vma). + * + * Difference from the generic handlers: CSRRC with rs1=x0 and + * CSRRSI/CSRRCI with uimm=0 must not write the CSR (privileged spec + * §2.2), so a read of a read-only CSR through them is legal. The generic + * csrrc/csrrsi/csrrci treat every access as a write. Same fix as the v1 + * core header (cpu/iss/include/cores/cv32e40p/priv.hpp). + */ + +#pragma once + +static inline void csr_decode(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + // In case traces are active, convert the CSR number into a name +#ifdef VP_TRACE_ACTIVE + insn->args[2].flags = (iss_decoder_arg_flag_e)(insn->args[2].flags | ISS_DECODER_ARG_FLAG_DUMP_NAME); + insn->args[2].name = iss_csr_name(iss, UIM_GET(0)); +#endif +} + +static inline iss_reg_t csrrw_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr) + { + return csr->handle(iss, insn, pc, REG_GET(0)); + } + + iss_reg_t value; + iss_reg_t reg_value = REG_GET(0); + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + iss_csr_write(iss, insn, UIM_GET(0), reg_value); + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrc_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + iss_reg_t reg_value = REG_GET(0); + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, REG_IN(0) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (REG_IN(0) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value & ~reg_value); + } + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrs_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + iss_reg_t reg_value = REG_GET(0); + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, REG_IN(0) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (REG_IN(0) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value | reg_value); + } + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrwi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr) + { + return csr->handle(iss, insn, pc, UIM_GET(1)); + } + + iss_reg_t value; + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + iss_csr_write(iss, insn, UIM_GET(0), UIM_GET(1)); + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrci_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, UIM_GET(1) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (UIM_GET(1) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value & ~UIM_GET(1)); + } + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t csrrsi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss_reg_t value; + + CsrAbtractReg *csr = iss->csr.get_csr(UIM_GET(0)); + if (csr && !csr->check_access(iss, UIM_GET(1) != 0, true)) + { + return pc; + } + + if (iss_csr_read(iss, insn, UIM_GET(0), &value) == 0) + { + if (insn->out_regs[0] != 0) + { + iss->regfile.memcheck_set_valid(REG_OUT(0), true); + REG_SET(0, value); + } + } + + if (UIM_GET(1) != 0) + { + iss_csr_write(iss, insn, UIM_GET(0), value | UIM_GET(1)); + } + + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t wfi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->irq.wfi_handle(insn); + return iss_insn_next(iss, insn, pc); +} + +static inline iss_reg_t mret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->exec.irq_exit.set(1); + iss->timing.stall_insn_dependency_account(5); + return iss->core.mret_handle(); +} + +static inline iss_reg_t dret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + return iss->core.dret_handle(); +} + +static inline iss_reg_t sret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + iss->timing.stall_insn_dependency_account(5); + return iss->core.sret_handle(); +} + +static inline iss_reg_t sfence_vma_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) +{ + if (iss->core.mode_get() == PRIV_S && iss->csr.mstatus.tvm) + { + iss->exception.raise(pc, ISS_EXCEPT_ILLEGAL); + return pc; + } + else + { +#ifdef CONFIG_GVSOC_ISS_MMU + iss->mmu.flush(REG_GET(0), REG_GET(1)); +#endif + iss->insn_cache.mode_flush(); + return iss_insn_next(iss, insn, pc); + } +} diff --git a/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp b/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp new file mode 100644 index 00000000..4f4b9c0e --- /dev/null +++ b/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp @@ -0,0 +1,56 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#pragma once + +#include +#include + +/* CV32E40P register-file personality. + * + * A trapped instruction writes no destination register on the RTL. The + * reserved-rounding-mode raise (isa_lib int.h) fires inside the value + * expression of the write-back macro, so the write that follows must be + * dropped: the raise site arms wb_suppress and the next set_reg/set_freg + * consumes it (one-shot — the instruction's own write always follows the + * raise within the same handler, so unrelated writes are never dropped). */ +class Cv32e40pRegfile : public Regfile +{ +public: + Cv32e40pRegfile(Iss &iss) : Regfile(iss) {} + + void reset(bool active) + { + this->wb_suppress = false; + this->Regfile::reset(active); + } + + inline void wb_suppress_arm() { this->wb_suppress = true; } + + /* Shadows the base setters (static dispatch via CONFIG_GVSOC_ISS_REGFILE). */ + inline void set_reg(int reg, uint64_t value) + { + if (this->wb_suppress) + { + this->wb_suppress = false; + return; + } + this->Regfile::set_reg(reg, value); + } + + inline void set_freg(int reg, uint64_t value) + { + if (this->wb_suppress) + { + this->wb_suppress = false; + return; + } + this->Regfile::set_freg(reg, value); + } + +private: + bool wb_suppress = false; +}; diff --git a/cpu/iss_v2/src/cores/cv32e40p/core.cpp b/cpu/iss_v2/src/cores/cv32e40p/core.cpp new file mode 100644 index 00000000..4f2c7a03 --- /dev/null +++ b/cpu/iss_v2/src/cores/cv32e40p/core.cpp @@ -0,0 +1,17 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#include + +iss_reg_t Cv32e40pCore::mret_handle() +{ + /* mcause holds its value across MRET on CV32E40P (cleared only by the + * next trap or an explicit CSR write); the generic handler zeroes it. */ + iss_reg_t mcause = this->iss.csr.mcause.value; + iss_reg_t pc = this->Core::mret_handle(); + this->iss.csr.mcause.value = mcause; + return pc; +} diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index 664c4c52..f2cdde5a 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -14,6 +14,7 @@ * which this core configures to raise illegal-instruction. */ #include +#include bool Cv32e40pRoCsr::check_access(Iss *iss, bool write, bool read) { @@ -45,6 +46,16 @@ bool Cv32e40pHwloopCsr::check_access(Iss *iss, bool write, bool read) return true; } +bool Cv32e40pCounterAlias::check_access(Iss *iss, bool write, bool read) +{ + if (write) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return false; + } + return true; +} + Cv32e40pCsr::Cv32e40pCsr(Iss &iss) : Csr(iss) { @@ -61,7 +72,8 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) 0x740, 0x741, 0x742, 0x744, /* mnscratch, mnepc, mncause, mnstatus */ 0x008, 0x009, 0x00A, 0x00F, /* vstart, vxstat, vxrm, vcsr */ 0xC20, 0xC21, 0xC22, /* vl, vtype, vlenb */ - 0xC00, 0xC01, 0xC02, /* cycle, time, instret */ + 0xC00, 0xC01, 0xC02, /* cycle, time, instret (0xC00/0xC02 + re-declared below as RO aliases) */ }; for (iss_reg_t addr : nonexistent) { @@ -124,13 +136,52 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) 0x323 + i, 0, (i < num_hpm) ? 0xFFFF : 0); } + /* User counter aliases: the RTL read mux maps 0xC00..0xC1F and + * 0xC80..0xC9F straight onto the machine counter bank + * (cv32e40p_cs_registers.sv); writes trap through check_access. */ + this->declare_csr(&this->cycle_alias, "cycle", 0xC00); + this->cycle_alias.register_callback(std::bind(&Cv32e40pCsr::cycle_alias_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->instret_alias, "instret", 0xC02); + this->instret_alias.register_callback(std::bind(&Cv32e40pCsr::instret_alias_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +#if ISS_REG_WIDTH == 32 + this->declare_csr(&this->cycleh_alias, "cycleh", 0xC80); + this->cycleh_alias.register_callback(std::bind(&Cv32e40pCsr::cycleh_alias_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->instreth_alias, "instreth", 0xC82); + this->instreth_alias.register_callback(std::bind(&Cv32e40pCsr::instreth_alias_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +#endif + for (int i = 0; i < 29; i++) + { + this->declare_csr(&this->hpmcounter_alias[i], "hpmcounter" + std::to_string(i + 3), + 0xC03 + i); + this->hpmcounter_alias[i].register_callback(std::bind(&Cv32e40pCsr::hpm_alias_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3, i)); +#if ISS_REG_WIDTH == 32 + this->declare_csr(&this->hpmcounterh_alias[i], "hpmcounter" + std::to_string(i + 3) + "h", + 0xC83 + i); + this->hpmcounterh_alias[i].register_callback(std::bind(&Cv32e40pCsr::hpmh_alias_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3, i)); +#endif + } + /* Interrupt and trap CSRs: masks from the RTL (cv32e40p_cs_registers.sv). * mstatus: only MIE/MPIE (and FS with an FPU) are writable; the mask is * applied by Core::mstatus_update via CONFIG_GVSOC_ISS_CORE_MSTATUS_WRITE_MASK * set by the recipe. Reset is MPP=M, FS=Off for every configuration. */ this->mstatus.reset_val = 0x00001800; - this->mie.set_write_mask(0xFFFF0888); - this->mip.set_write_mask(0); + /* mie: declarative only — IrqRiscv::mie_access bypasses the register + * write mask; the effective masking is Cv32e40pIrq::mie_write_fixup. */ + this->mie.set_write_mask(Cv32e40pIrq::IRQ_MASK); + /* mip: read-only front-end replaces the base register in the CSR map + * (see csr.hpp). The base object stays as the wire-driven store, its + * IrqRiscv wire callbacks keep writing it directly. */ + this->undeclare_csr(0x344); + this->declare_csr(&this->mip_view, "mip", 0x344); + this->mip_view.register_callback(std::bind(&Cv32e40pCsr::mip_view_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); this->mtvec.set_write_mask(0xFFFFFF01); this->mtvec.reset_val = 0x1; this->mtval.set_write_mask(0); @@ -220,6 +271,13 @@ void Cv32e40pCsr::reset(bool active) this->dcsr = (4 << 28) | 0x3; this->mcycle_offset = 0; + + /* Excluded from the base reset sweep, which walks the CSR map: + * mip left it for the mip_view front-end, hwloop_lpend is a plain + * shadow. */ + this->mip.value = 0; + this->hwloop_lpend[0] = 0; + this->hwloop_lpend[1] = 0; } } @@ -265,6 +323,16 @@ bool Cv32e40pCsr::tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t & return false; } +bool Cv32e40pCsr::mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + /* Reads mirror the wire-driven register; writes are dropped. */ + if (!is_write) + { + value = this->mip.value; + } + return false; +} + uint64_t Cv32e40pCsr::mcycle_count() { #if ISS_REG_WIDTH == 32 @@ -316,6 +384,61 @@ bool Cv32e40pCsr::mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &val return false; } +void Cv32e40pCsr::fp_state_dirty() +{ +#if CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA + /* A trapped FP instruction has no architectural side effects; the raise + * (reserved rounding mode, isa_lib int.h) runs inside the write-back + * macro before this call. */ + if (this->iss.exec.has_exception) + return; + this->mstatus.fs = 3; +#endif +} + +/* User counter aliases: read-only mirrors of the machine counters (writes + * never reach these callbacks, Cv32e40pCounterAlias::check_access traps + * them first). */ +bool Cv32e40pCsr::cycle_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + value = (iss_reg_t)this->mcycle_count(); + return false; +} + +bool Cv32e40pCsr::cycleh_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + value = (iss_reg_t)(this->mcycle_count() >> 32); + return false; +} + +bool Cv32e40pCsr::instret_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + value = this->minstret.value; + return false; +} + +bool Cv32e40pCsr::instreth_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ +#if ISS_REG_WIDTH == 32 + value = this->minstreth.value; +#endif + return false; +} + +bool Cv32e40pCsr::hpm_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index) +{ + value = this->mhpmcounter[index].value; + return false; +} + +bool Cv32e40pCsr::hpmh_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index) +{ +#if ISS_REG_WIDTH == 32 + value = this->mhpmcounterh[index].value; +#endif + return false; +} + bool Cv32e40pCsr::mcountinhibit_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) { /* Freeze the count into the register pair when CY sets, re-anchor the @@ -354,9 +477,9 @@ bool Cv32e40pCsr::hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t & switch (index % 3) { case 0: value = this->iss.hwloop.get_start(loop); break; - /* The Hwloop module stores the loop-back point, LPEND - 4 (see the - * corev.hpp setters); the architectural LPEND is re-derived here. */ - case 1: value = this->iss.hwloop.get_end(loop) + 4; break; + /* The Hwloop module stores the loop-back point, LPEND - 4; the + * architectural value lives in the shadow the setters keep. */ + case 1: value = this->hwloop_lpend[loop]; break; case 2: value = this->iss.hwloop.get_count(loop); break; } return false; diff --git a/cpu/iss_v2/src/cores/cv32e40p/exception.cpp b/cpu/iss_v2/src/cores/cv32e40p/exception.cpp new file mode 100644 index 00000000..18ac5c2a --- /dev/null +++ b/cpu/iss_v2/src/cores/cv32e40p/exception.cpp @@ -0,0 +1,24 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +#include + +void Cv32e40pException::raise(iss_reg_t pc, int id) +{ + /* Commit-stream consumers gate state compares on this (events.hpp). */ + this->iss.timing.trap_seq++; + + this->Exception::raise(pc, id); + + /* Exceptions enter at the mtvec base. mtvec.value keeps the RTL mode + * bits (mtvec[1:0] = 01) and the generic raise copies it verbatim + * into the entry PC. Assumes the mtvec-based entry path, i.e. + * CONFIG_GVSOC_ISS_RISCV_EXCEPTIONS (always set by Cv32e40pIrq). */ + if (id != ISS_EXCEPT_DEBUG) + { + this->iss.exec.exception_pc &= ~(iss_reg_t)0x3; + } +} diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp new file mode 100644 index 00000000..865ec533 --- /dev/null +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -0,0 +1,104 @@ +// SPDX-FileCopyrightText: 2026 Fondazione Chips-it +// +// SPDX-License-Identifier: Apache-2.0 +// +// Authors: Marco Paci (marco.paci@chips.it) + +/* CV32E40P interrupt personality for iss_v2. + * + * The generic IrqRiscv take (irq_riscv.cpp check()) uses the standard + * RISC-V priority ladder and jumps to the mtvec base for every interrupt. + * The RTL differs on both counts (cv32e40p_int_controller.sv, + * cv32e40p_controller.sv): the fast lines irq[31:16] outrank MEI/MSI/MTI, + * and vectored mode sends each interrupt to base + 4*id. This override + * implements the RTL behaviour; delivery (mip update, WFI wake-up) stays + * on the inherited wire-sync path. */ + +#include + +void Cv32e40pIrq::start() +{ + /* Registered here so it runs after IrqRiscv::mie_access (bound in the + * base constructor), which stores the written value unmasked and + * suppresses the register's own write mask. */ + this->iss.csr.mie.register_callback(std::bind(&Cv32e40pIrq::mie_write_fixup, + this, std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +} + +/* RTL WARL result: only the wired interrupt lines are writable in mie + * (cv32e40p_cs_registers.sv, csr_mie_wdata & IRQ_MASK). */ +bool Cv32e40pIrq::mie_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (!is_write) + { + return true; + } + this->iss.csr.mie.value &= IRQ_MASK; + return false; +} + +/* RTL priority: irq[31] highest, down to irq[16], then MEI(11), MSI(3), + * MTI(7). The caller guarantees at least one bit of IRQ_MASK is set. */ +static int cv32e40p_irq_pick(iss_reg_t pending) +{ + for (int id = 31; id >= 16; id--) + { + if ((pending >> id) & 1) + { + return id; + } + } + if ((pending >> 11) & 1) return 11; + if ((pending >> 3) & 1) return 3; + return 7; +} + +int Cv32e40pIrq::check() +{ + /* Debug entry: same as the generic implementation. */ + if (this->req_debug && !this->iss.exec.debug_mode) + { + this->iss.exec.debug_mode = true; + this->iss.csr.depc = this->iss.exec.current_insn; + this->debug_saved_irq_enable = this->irq_enable.get(); + this->irq_enable.set(0); + this->req_debug = false; + this->iss.exec.current_insn = this->debug_handler; + return 1; + } + + /* M-mode only core: the take needs a wired pending line and the global + * enable (mstatus.MIE). */ + iss_reg_t pending = this->iss.csr.mie.value & this->iss.csr.mip.value & IRQ_MASK; + if (!pending || !this->iss.csr.mstatus.mie) + { + return 0; + } + + int irq = cv32e40p_irq_pick(pending); + + /* mtvec holds {base[31:8], 0, mode} (Cv32e40pCsr::mtvec_write_fixup); + * vectored mode enters at base + 4*id, direct mode at base. */ + iss_reg_t base = this->iss.csr.mtvec.value & 0xFFFFFF00; + iss_reg_t entry = (this->iss.csr.mtvec.value & 1) ? base + (irq << 2) : base; + + this->trace.msg(vp::Trace::LEVEL_TRACE, "Handling IRQ (irq: %d, entry: 0x%lx)\n", + irq, entry); + + /* Commit-stream consumers gate state compares on this (events.hpp). */ + this->iss.timing.trap_seq++; + + this->iss.exec.interrupt_taken(); + this->iss.csr.mepc.value = this->iss.exec.current_insn; + this->iss.csr.mstatus.mpie = this->iss.csr.mstatus.mie; + this->iss.csr.mstatus.mie = 0; + this->iss.csr.mstatus.mpp = this->iss.core.mode_get(); + this->iss.csr.mcause.value = (1ULL << (ISS_REG_WIDTH - 1)) | (unsigned int)irq; + this->iss.exec.current_insn = entry; + this->iss.core.mode_set(PRIV_M); + this->irq_enable.set(0); + + this->iss.timing.stall_insn_dependency_account(4); + + return 1; +} diff --git a/cv32e40p-v2-standalone-nopulp.py b/cv32e40p-v2-standalone-nopulp.py new file mode 100644 index 00000000..e63d0537 --- /dev/null +++ b/cv32e40p-v2-standalone-nopulp.py @@ -0,0 +1,37 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P iss_v2 co-simulation target, no-PULP configuration (rv32imc, no CoreV extensions). + +import gvsoc.runner +from pulp.cv32e40p_v2_standalone import Cv32e40pStandaloneTop + + +class Cv32e40p(Cv32e40pStandaloneTop): + + def __init__(self, parent, name=None): + super().__init__(parent, name, pulp=0) + + +class Target(gvsoc.runner.Target): + + description = "CV32E40P iss_v2 standalone for UVM co-simulation (no PULP)" + model = Cv32e40p + name = "cv32e40p-v2-standalone-nopulp" diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index a6f9a93d..5f29222d 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -23,7 +23,7 @@ from typing import Iterable from typing_extensions import override from gvsoc.systree import Component -from cpu.iss_v2.riscv import (RiscvCommon, IssModule, Irq, ExecInOrder, +from cpu.iss_v2.riscv import (RiscvCommon, IssModule, ExecInOrder, Regfile, LsuV2, Hwloop) from cpu.iss.isa_gen.isa_gen import Isa, IsaSubset from cpu.iss.isa_gen.isa_riscv_gen import RiscvIsa @@ -38,6 +38,34 @@ _MISA_BASE = 0x40001104 +def _apply_rtl_decode_fixes(isa: Isa) -> None: + """Decode-level differences between the generated RISC-V tables and the + CV32E40P RTL, applied once per ISA instance: + + - FENCE/FENCE.I: the RTL decoder checks funct3 only and ignores the + reserved rd/rs1/fm fields (cv32e40p_decoder.sv, OPCODE_FENCE), as the + unprivileged spec requires for forward compatibility; the generated + encodings pin those fields to zero, turning e.g. a fence with rd=x31 + into an illegal instruction. + - CSRRC with rs1=x0 and CSRRSI/CSRRCI with uimm=0 must not write the CSR + (privileged spec §2.2): the priv subset is routed to the core's + handlers (cores/cv32e40p/priv.hpp), same fix as the v1 model. + """ + relaxed = { + 'fence': '------- ----- ----- 000 ----- 0001111', + 'fence.i': '------- ----- ----- 001 ----- 0001111', + } + for insn in isa.get_isa('rv32i').instrs: + encoding = relaxed.get(insn.label) + if encoding is not None: + # Same transform as Instr.__init__ (reversed, spaces stripped). + insn.encoding = encoding[::-1].replace(' ', '') + + isa.get_isa('priv').includes = [ + '', + ] + + class Cv32e40pConfig(RiscvConfig): pass @@ -58,6 +86,26 @@ def gen(self, iss: RiscvCommon): iss.isa.add_implem_include('') +class Cv32e40pIrq(IssModule): + """CV32E40P interrupt personality. + + Selects the Cv32e40pIrq C++ class for the irq slot: RISC-V privileged + interrupt scheme (mie/mip/mtvec, standard mcause codes) with the RTL + priority order (fast lines irq[31:16] above MEI/MSI/MTI) and vectored + entry (base + 4*id). + """ + + @override + def gen(self, iss: RiscvCommon): + iss.isa.add_define('CONFIG_GVSOC_ISS_IRQ', 'Cv32e40pIrq') + iss.isa.add_define('CONFIG_GVSOC_ISS_RISCV_EXCEPTIONS', 1) + iss.isa.add_include('') + iss.add_sources([ + 'cpu/iss_v2/src/irq/irq_riscv.cpp', + 'cpu/iss_v2/src/cores/cv32e40p/irq.cpp', + ]) + + class Cv32e40pEvent(IssModule): """CV32E40P event accounting. @@ -101,6 +149,10 @@ def gen(self, iss: RiscvCommon): # FP write-backs must dirty mstatus.FS (see iss_v2 isa_lib/macros.h). iss.isa.add_define('CONFIG_GVSOC_ISS_FP_STATE_DIRTY', 1) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_ZFINX', 1 if self.zfinx else 0) + if self.fpu or self.zfinx: + # Reserved FP rounding modes raise illegal-instruction with no + # architectural side effects (isa_lib int.h + Cv32e40pRegfile). + iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_FP_TRAPS', 1) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_PULP', 1 if self.pulp else 0) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS', self.num_mhpmcounters) # mstatus write policy, applied by Core::mstatus_update: only @@ -113,6 +165,58 @@ def gen(self, iss: RiscvCommon): ]) +class Cv32e40pRegfileModule(IssModule): + """CV32E40P register-file personality. + + Selects the Cv32e40pRegfile C++ class for the regfile slot: a trapped + instruction writes no destination register (the reserved-rounding-mode + raise in isa_lib int.h arms the one-shot write-back suppression). + """ + + @override + def gen(self, iss: RiscvCommon): + iss.isa.add_define('CONFIG_GVSOC_ISS_REGFILE', 'Cv32e40pRegfile') + iss.isa.add_define('CONFIG_GVSOC_ISS_REGFILE_SCOREBOARD', '1') + iss.isa.add_include('') + iss.add_sources(['cpu/iss_v2/src/regfile.cpp']) + + +class Cv32e40pCoreModule(IssModule): + """CV32E40P core personality. + + Selects the Cv32e40pCore C++ class for the core slot: MRET keeps + mcause (the RTL holds it until the next trap or an explicit CSR + write, the generic handler clears it). + """ + + @override + def gen(self, iss: RiscvCommon): + iss.isa.add_define('CONFIG_GVSOC_ISS_CORE', 'Cv32e40pCore') + iss.isa.add_include('') + iss.add_sources([ + 'cpu/iss_v2/src/core.cpp', + 'cpu/iss_v2/src/cores/cv32e40p/core.cpp', + ]) + + +class Cv32e40pExceptionModule(IssModule): + """CV32E40P exception personality. + + Selects the Cv32e40pException C++ class for the exception slot: + exceptions enter at the mtvec base, with the mode bits kept out of + the entry PC. + """ + + @override + def gen(self, iss: RiscvCommon): + iss.isa.add_define('CONFIG_GVSOC_ISS_EXCEPTION', 'Cv32e40pException') + iss.isa.add_include('') + iss.add_sources([ + 'cpu/iss_v2/src/exception.cpp', + 'cpu/iss_v2/src/cores/cv32e40p/exception.cpp', + ]) + + class Cv32e40p(RiscvCommon): """CV32E40P on the iss_v2 modular core. @@ -153,6 +257,8 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, # FPU == 1 && ZFINX == 0 (cv32e40p_compressed_decoder.sv). isa_instance.disable_from_isa_tag('cf') + _apply_rtl_decode_fixes(isa_instance) + isa_instances[cache_key] = isa_instance misa = _MISA_BASE @@ -162,16 +268,15 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, misa |= 1 << 23 # X modules: dict[str, IssModule] = { - # RISC-V privileged interrupt/exception scheme (mie/mip/mtvec, - # standard mcause codes), as implemented by the RTL. The PULP - # event-unit style IrqExternal does not apply to this core. - 'irq': Irq(), + 'irq': Cv32e40pIrq(), + 'core': Cv32e40pCoreModule(), + 'exception': Cv32e40pExceptionModule(), 'event': Cv32e40pEvent(), 'csr': Cv32e40pCsr(fpu=fpu, zfinx=zfinx, pulp=pulp, num_mhpmcounters=num_mhpmcounters), 'exec': Cv32e40pExec(), 'lsu': LsuV2(), - 'regfile': Regfile(scoreboard=True), + 'regfile': Cv32e40pRegfileModule(), 'hwloop': Hwloop(), } diff --git a/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp b/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp new file mode 100644 index 00000000..67f6e51a --- /dev/null +++ b/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp @@ -0,0 +1,108 @@ +/* + * Copyright (C) 2026 Fondazione Chips-it + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) + */ + +/* + * CV32E40P interrupt-line injector. + * + * Exposes the core interrupt wires to an external gv:: client: the + * co-simulation bridge binds each line with gv::wire_bind and drives it + * as the RTL irq_i inputs change. Every line forwards to the matching + * IrqRiscv slave port, so mip and the wake-up logic follow the same path + * as a platform interrupt source. + */ + +#include + +#include +#include +#include + +class Cv32e40pIrqInjector : public vp::Component +{ +public: + Cv32e40pIrqInjector(vp::ComponentConf &config); + + void *external_bind(std::string comp_name, std::string itf_name, + void *handle) override; + +private: + /* One interrupt line: gv::Wire_binding facade over the master port. */ + class Line : public gv::Wire_binding + { + public: + void update(int value) override + { + if (this->itf.is_bound()) + { + this->itf.sync(value != 0); + } + } + + std::string name; + vp::WireMaster itf; + }; + + /* msi, mti, mei plus the sixteen fast lines irq[31:16]. */ + static constexpr int NB_LINES = 19; + Line lines[NB_LINES]; +}; + +Cv32e40pIrqInjector::Cv32e40pIrqInjector(vp::ComponentConf &config) + : vp::Component(config) +{ + this->lines[0].name = "msi"; + this->lines[1].name = "mti"; + this->lines[2].name = "mei"; + for (int i = 3; i < NB_LINES; i++) + { + this->lines[i].name = "external_irq_" + std::to_string(16 + i - 3); + } + + for (auto &line : this->lines) + { + this->new_master_port(line.name, &line.itf); + } +} + +void *Cv32e40pIrqInjector::external_bind(std::string comp_name, + std::string itf_name, void *handle) +{ + (void)handle; + + if (comp_name != this->get_name()) + { + return NULL; + } + + for (auto &line : this->lines) + { + if (line.name == itf_name) + { + return static_cast(&line); + } + } + + return NULL; +} + +extern "C" vp::Component *gv_new(vp::ComponentConf &config) +{ + return new Cv32e40pIrqInjector(config); +} diff --git a/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.py b/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.py new file mode 100644 index 00000000..45768a90 --- /dev/null +++ b/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.py @@ -0,0 +1,42 @@ +# +# Copyright (C) 2026 Fondazione Chips-it +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# + +# +# Authors: Marco Paci, Fondazione Chips-it (marco.paci@chips.it) +# + +# CV32E40P interrupt-line injector: lets the external co-simulation bridge +# drive the core interrupt wires (msi/mti/mei + fast irq16..31) through +# gv::wire_bind. One master wire port per line, named after the core slave +# port it drives. + +import gvsoc.systree + +# Line names and their interrupt numbers, index-aligned, in RVVI net order +# (MSWInterrupt, MTimerInterrupt, MExternalInterrupt, LocalInterrupt0..15). +IRQ_LINES: tuple = ('msi', 'mti', 'mei', + *(f'external_irq_{i}' for i in range(16, 32))) +IRQ_NUMBERS: tuple = (3, 7, 11, *range(16, 32)) + + +class Cv32e40pIrqInjector(gvsoc.systree.Component): + + def __init__(self, parent, name): + super().__init__(parent, name) + self.add_sources(['pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp']) + + def o_LINE(self, name: str, itf: gvsoc.systree.SlaveItf): + self.itf_bind(name, itf, signature='wire') diff --git a/pulp/cv32e40p_v2_standalone.py b/pulp/cv32e40p_v2_standalone.py index 75da31ba..2655e03e 100644 --- a/pulp/cv32e40p_v2_standalone.py +++ b/pulp/cv32e40p_v2_standalone.py @@ -40,6 +40,8 @@ from utils.loader.loader_v2 import ElfLoader from pulp.cv32e40p_exit.cv32e40p_exit_device_v2 import Cv32e40pExitDeviceV2 from pulp.cv32e40p_sparse_mem.cv32e40p_sparse_mem_v2 import Cv32e40pSparseMemV2 +from pulp.cv32e40p_irq_injector.cv32e40p_irq_injector import (Cv32e40pIrqInjector, + IRQ_LINES, IRQ_NUMBERS) from pulp.cpu.iss.cv32e40p_v2 import Cv32e40p as Cv32e40pV2Core, Cv32e40pConfig @@ -58,6 +60,10 @@ class Cv32e40pStandaloneConfig(Config): "ZFINX configuration (FP operations on the integer register file)" )) + pulp: int = cfg_field(default=1, dump=True, desc=( + "PULP configuration (CoreV extensions, misa X bit)" + )) + core: Cv32e40pConfig = cfg_field(init=False, desc=( "CV32E40P core configuration" )) @@ -153,7 +159,8 @@ def __init__(self, parent, name, config: Cv32e40pStandaloneConfig, binary): bg_mem = Cv32e40pSparseMemV2 ( self, 'background_mem' ) ico = Router ( self, 'ico' , config=config.router ) core = Cv32e40pV2Core ( self, 'core' , config=config.core , - fpu=bool(config.fpu), zfinx=bool(config.zfinx) ) + fpu=bool(config.fpu), zfinx=bool(config.zfinx), + pulp=bool(config.pulp) ) loader = ElfLoader ( self, 'loader' , binary=binary ) ico.o_MAP ( mem.i_INPUT() , mapping=config.mem_mapping ) @@ -173,17 +180,24 @@ def __init__(self, parent, name, config: Cv32e40pStandaloneConfig, binary): core.o_FETCH ( ico.i_INPUT(1) ) core.o_DATA ( ico.i_INPUT(2) ) + # Interrupt lines: the co-sim bridge drives the injector through + # gv::wire_bind; each line lands on the core's native slave port, + # so mip and the WFI wake-up follow the hardware path. + irq_inj = Cv32e40pIrqInjector(self, 'irq_injector') + for name, irq in zip(IRQ_LINES, IRQ_NUMBERS): + irq_inj.o_LINE(name, core.i_IRQ(irq)) + class Cv32e40pStandaloneTop(gvsoc.systree.Component): - def __init__(self, parent, name=None, fpu: int=0, zfinx: int=0): + def __init__(self, parent, name=None, fpu: int=0, zfinx: int=0, pulp: int=1): super().__init__(parent, name) binary = TargetParameter( self, name='binary', value=None, description='ELF binary to simulate' ).get_value() - config = Cv32e40pStandaloneConfig('soc', fpu=fpu, zfinx=zfinx) + config = Cv32e40pStandaloneConfig('soc', fpu=fpu, zfinx=zfinx, pulp=pulp) clock = vp.clock_domain.Clock_domain(self, 'clock', frequency=50000000) soc = Cv32e40pStandaloneSoc(self, 'soc', config, binary) From b305ee76d548eb65cd968df86e796b4b728ca75a Mon Sep 17 00:00:00 2001 From: mpaci Date: Sun, 19 Jul 2026 21:55:25 +0200 Subject: [PATCH 12/28] fix(cv32e40p): SIMD operand order, atomic debug entry and drain accessor on the v2 core - The recipe now emits RISCV into the generated ISA header: the shared isa_lib int.h selects the RISC-V (not legacy RISCY) source order for lib_VEC_SHUFFLE2_*, and no v2 build was defining it (the v1 build gets it from the iss CMakeLists). Fixes cv.shuffle2.h half-word swap. - Debug entry writes dcsr.cause atomically with the entry, as the RTL does, and bumps trap_seq so a commit-FIFO entry spanning a haltreq redirect is flagged stale like any other trap redirect. - inflight_pending() exposes the parked-instruction state the bridge needs to drain the pipeline before a forced redirect. --- cpu/iss_v2/include/cores/cv32e40p/events.hpp | 5 +++++ cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 7 ++++++- pulp/cpu/iss/cv32e40p_v2.py | 4 ++++ 3 files changed, 15 insertions(+), 1 deletion(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/events.hpp b/cpu/iss_v2/include/cores/cv32e40p/events.hpp index 34489a97..fdd825f6 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events.hpp @@ -59,6 +59,11 @@ class Cv32e40pEvents : public Events /* Drop commits not consumed yet (external resync forced a new PC). */ inline void commit_stream_flush(); + /* Instructions parked in the exec commit FIFO, not yet drained. Their + * load-use scoreboard bits are already set, so redirecting while this + * is true needs a drain first (see gvsoc_engine_set_pc). */ + bool inflight_pending() const { return this->inflight_pop != this->inflight_push; } + private: /* Program-order PCs of the instructions parked in the exec commit * FIFO (held or sync follower); drain pops them in the same order. diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index 865ec533..27f5c0df 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -55,11 +55,16 @@ static int cv32e40p_irq_pick(iss_reg_t pending) int Cv32e40pIrq::check() { - /* Debug entry: same as the generic implementation. */ + /* Debug entry: generic implementation plus dcsr.cause, written + * atomically with the entry as the RTL does (the bridge only raises + * req_debug). Only haltreq (cause=3) can get here in co-simulation. */ if (this->req_debug && !this->iss.exec.debug_mode) { this->iss.exec.debug_mode = true; this->iss.csr.depc = this->iss.exec.current_insn; + this->iss.csr.dcsr = (this->iss.csr.dcsr & ~(0x7u << 6)) | (3u << 6); + /* Commit-stream consumers gate state compares on this (events.hpp). */ + this->iss.timing.trap_seq++; this->debug_saved_irq_enable = this->irq_enable.get(); this->irq_enable.set(0); this->req_debug = false; diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index 5f29222d..51042d9a 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -144,6 +144,10 @@ def __init__(self, fpu: bool=False, zfinx: bool=False, pulp: bool=True, def gen(self, iss: RiscvCommon): iss.isa.add_define('CONFIG_GVSOC_ISS_CSR', 'Cv32e40pCsr') iss.isa.add_include('') + # Select the RISC-V (not legacy RISCY) SIMD operand order in the + # shared isa_lib int.h (lib_VEC_SHUFFLE2_*); the v1 build gets this + # from the iss CMakeLists. + iss.isa.add_define('RISCV', 1) iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_FPU_IN_ISA', 1 if self.fpu else 0) if self.fpu: # FP write-backs must dirty mstatus.FS (see iss_v2 isa_lib/macros.h). From eeda65dba4184a460edd75721bc3c16cbae2b4f9 Mon Sep 17 00:00:00 2001 From: mpaci Date: Mon, 20 Jul 2026 21:42:44 +0200 Subject: [PATCH 13/28] feat(cv32e40p): debug-mode support on the v2 core Debug entry with a caller-supplied cause (req_debug_cause, written into dcsr atomically with the entry), debug-CSR views over the base raw fields (dcsr with the RTL WARL mask, dpc 16-bit aligned, dscratch0/1) so the debug-ROM csrrw accesses stop raising illegal-instruction, no interrupt take while in debug mode (the RTL controller ignores irq_req there), and the debug_handler address in the v2 recipe (was 0). --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 15 +++++ cpu/iss_v2/include/cores/cv32e40p/irq.hpp | 7 +++ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 76 +++++++++++++++++++++++ cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 19 +++++- pulp/cpu/iss/cv32e40p_v2.py | 5 +- 5 files changed, 118 insertions(+), 4 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index 725ac23b..8f757cf2 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -137,6 +137,17 @@ class Cv32e40pCsr : public Csr * never corrupt the pending state. */ CsrAbtractReg mip_view; + /* Debug-mode CSRs (0x7B0-0x7B3): views over the base raw fields + * (dcsr/depc/scratch0/scratch1), which the debug-entry and dret paths + * write directly. The base register file leaves these addresses + * undeclared, so without the views every debug-ROM csrrw raises + * illegal-instruction. The RTL has no debug-mode access gate + * (cv32e40p_cs_registers.sv decodes them at any time). */ + CsrAbtractReg dcsr_view; /* 0x7B0 */ + CsrAbtractReg dpc_view; /* 0x7B1 */ + CsrAbtractReg dscratch0_view; /* 0x7B2 */ + CsrAbtractReg dscratch1_view; /* 0x7B3 */ + /* User counter aliases: 0xC00/0xC02/0xC03..0xC1F and the H views at * 0xC80/0xC82/0xC83..0xC9F. time (0xC01) is absent. */ Cv32e40pCounterAlias cycle_alias; @@ -160,6 +171,10 @@ class Cv32e40pCsr : public Csr bool hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); bool tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool dcsr_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool dpc_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool dscratch0_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool dscratch1_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool cycle_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); diff --git a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp index f1477643..28e12d6d 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp @@ -25,6 +25,13 @@ class Cv32e40pIrq : public IrqRiscv * CONFIG_GVSOC_ISS_IRQ). */ int check(); + /* dcsr.cause for the next req_debug take (debug spec: 1=ebreak, + * 3=haltreq, 4=single-step). The haltreq wire path leaves the + * default; the co-simulation bridge's informed debug entry sets it + * from the DUT's dcsr before arming req_debug. Reset to 3 by the + * entry itself. */ + int req_debug_cause = 3; + private: bool mie_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); }; diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index f2cdde5a..95e96bb1 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -200,6 +200,23 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) this->declare_csr(&this->mcontext, "mcontext", 0x7A8, 0, 0); this->declare_csr(&this->scontext, "scontext", 0x7AA, 0, 0); + /* Debug-mode CSRs: views over the base raw fields, kept coherent with + * the debug-entry (Cv32e40pIrq::check) and dret paths which write the + * raw fields directly. Undeclared in the base, they would otherwise + * raise illegal-instruction on the first debug-ROM access. */ + this->declare_csr(&this->dcsr_view, "dcsr", 0x7B0); + this->dcsr_view.register_callback(std::bind(&Cv32e40pCsr::dcsr_view_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->dpc_view, "dpc", 0x7B1); + this->dpc_view.register_callback(std::bind(&Cv32e40pCsr::dpc_view_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->dscratch0_view, "dscratch0", 0x7B2); + this->dscratch0_view.register_callback(std::bind(&Cv32e40pCsr::dscratch0_view_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->declare_csr(&this->dscratch1_view, "dscratch1", 0x7B3); + this->dscratch1_view.register_callback(std::bind(&Cv32e40pCsr::dscratch1_view_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + #if CONFIG_GVSOC_ISS_CV32E40P_PULP /* PULP custom CSRs. The RTL decoder raises illegal-instruction on any * write to them (read-only register type). */ @@ -333,6 +350,65 @@ bool Cv32e40pCsr::mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &va return false; } +/* RTL WARL (cv32e40p_cs_registers.sv, CSR_DCSR): writable bits are + * ebreakm(15), ebreaku(12), stepie(11), step(2) and prv[1:0] (WARL: M when + * written as M, U otherwise). xdebugver, cause and the hardwired-zero + * fields keep the stored value, which the debug entry writes directly. */ +bool Cv32e40pCsr::dcsr_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + constexpr iss_reg_t WRITABLE = (1u << 15) | (1u << 12) | (1u << 11) | (1u << 2); + iss_reg_t prv = ((value & 0x3) == 0x3) ? 0x3 : 0x0; + this->dcsr = (this->dcsr & ~(WRITABLE | 0x3)) | (value & WRITABLE) | prv; + } + else + { + value = this->dcsr; + } + return false; +} + +/* RTL forces 16-bit alignment on dpc writes (depc_n = wdata & ~1). */ +bool Cv32e40pCsr::dpc_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->depc = value & ~(iss_reg_t)1; + } + else + { + value = this->depc; + } + return false; +} + +bool Cv32e40pCsr::dscratch0_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->scratch0 = value; + } + else + { + value = this->scratch0; + } + return false; +} + +bool Cv32e40pCsr::dscratch1_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + if (is_write) + { + this->scratch1 = value; + } + else + { + value = this->scratch1; + } + return false; +} + uint64_t Cv32e40pCsr::mcycle_count() { #if ISS_REG_WIDTH == 32 diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index 27f5c0df..b9c1132c 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -56,13 +56,16 @@ static int cv32e40p_irq_pick(iss_reg_t pending) int Cv32e40pIrq::check() { /* Debug entry: generic implementation plus dcsr.cause, written - * atomically with the entry as the RTL does (the bridge only raises - * req_debug). Only haltreq (cause=3) can get here in co-simulation. */ + * atomically with the entry as the RTL does. The cause comes from + * req_debug_cause: 3 (haltreq) on the wire path, 1 (ebreak) or 4 + * (single-step) when the bridge's informed debug entry armed it. */ if (this->req_debug && !this->iss.exec.debug_mode) { this->iss.exec.debug_mode = true; this->iss.csr.depc = this->iss.exec.current_insn; - this->iss.csr.dcsr = (this->iss.csr.dcsr & ~(0x7u << 6)) | (3u << 6); + this->iss.csr.dcsr = (this->iss.csr.dcsr & ~(0x7u << 6)) | + ((iss_reg_t)(this->req_debug_cause & 0x7) << 6); + this->req_debug_cause = 3; /* Commit-stream consumers gate state compares on this (events.hpp). */ this->iss.timing.trap_seq++; this->debug_saved_irq_enable = this->irq_enable.get(); @@ -72,6 +75,16 @@ int Cv32e40pIrq::check() return 1; } + /* No interrupt is taken in debug mode (the RTL controller ignores + * irq_req entirely there). Explicit guard: inside the take_debug + * injection window the defense is down until the first debug-ROM + * commit lands, so check() can run again right after the entry and + * the ladder below would hijack it with a pending line. */ + if (this->iss.exec.debug_mode) + { + return 0; + } + /* M-mode only core: the take needs a wired pending line and the global * enable (mstatus.MIE). */ iss_reg_t pending = this->iss.csr.mie.value & this->iss.csr.mip.value & IRQ_MASK; diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index 51042d9a..f9597703 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -284,5 +284,8 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, 'hwloop': Hwloop(), } + # dm_halt_addr of the RTL testbench: the debug entry redirects here + # (linker script `dbg` region, loaded from the test ELF). super().__init__(parent, name, config=config, isa=isa_instance, - misa=misa, zfinx=zfinx, modules=modules) + misa=misa, zfinx=zfinx, modules=modules, + debug_handler=0x1A110800) From e2828d4fddf8b0e7e75cc011c6dea1a816ecd513 Mon Sep 17 00:00:00 2001 From: mpaci Date: Mon, 20 Jul 2026 23:30:35 +0200 Subject: [PATCH 14/28] fix(cv32e40p): reject M-mode access to the debug CSRs The 0x7B0-0x7B3 views were declared with the base CsrAbtractReg type, whose default check_access grants every access. The RTL decoder (cv32e40p_decoder.sv, CSR_DCSR..CSR_DSCRATCH1) raises illegal-instruction whenever they are touched outside debug mode, and generic_exception_test checks exactly that. Give the views a Cv32e40pDebugCsr type that raises illegal-instruction while exec.debug_mode is clear; debug-ROM code still passes the check. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 24 +++++++++++++++++------ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 10 ++++++++++ 2 files changed, 28 insertions(+), 6 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index 8f757cf2..82154fa9 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -71,6 +71,17 @@ class Cv32e40pCounterAlias : public CsrAbtractReg bool check_access(Iss *iss, bool write, bool read) override; }; +/* Debug-mode CSR front-end (dcsr/dpc/dscratch0-1): accessible only while in + * debug mode. The RTL decoder raises illegal-instruction on any M-mode + * access (cv32e40p_decoder.sv, CSR_DCSR..CSR_DSCRATCH1 with !debug_mode_i), + * so generic_exception_test relies on these accesses trapping. Debug-ROM + * code runs with debug_mode set and passes the check. */ +class Cv32e40pDebugCsr : public CsrAbtractReg +{ +public: + bool check_access(Iss *iss, bool write, bool read) override; +}; + class Cv32e40pCsr : public Csr { public: @@ -141,12 +152,13 @@ class Cv32e40pCsr : public Csr * (dcsr/depc/scratch0/scratch1), which the debug-entry and dret paths * write directly. The base register file leaves these addresses * undeclared, so without the views every debug-ROM csrrw raises - * illegal-instruction. The RTL has no debug-mode access gate - * (cv32e40p_cs_registers.sv decodes them at any time). */ - CsrAbtractReg dcsr_view; /* 0x7B0 */ - CsrAbtractReg dpc_view; /* 0x7B1 */ - CsrAbtractReg dscratch0_view; /* 0x7B2 */ - CsrAbtractReg dscratch1_view; /* 0x7B3 */ + * illegal-instruction. Access is legal from debug mode only: the RTL + * decoder (not cv32e40p_cs_registers.sv, which decodes them at any + * time) rejects M-mode accesses with illegal-instruction. */ + Cv32e40pDebugCsr dcsr_view; /* 0x7B0 */ + Cv32e40pDebugCsr dpc_view; /* 0x7B1 */ + Cv32e40pDebugCsr dscratch0_view; /* 0x7B2 */ + Cv32e40pDebugCsr dscratch1_view; /* 0x7B3 */ /* User counter aliases: 0xC00/0xC02/0xC03..0xC1F and the H views at * 0xC80/0xC82/0xC83..0xC9F. time (0xC01) is absent. */ diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index 95e96bb1..1d53ed88 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -56,6 +56,16 @@ bool Cv32e40pCounterAlias::check_access(Iss *iss, bool write, bool read) return true; } +bool Cv32e40pDebugCsr::check_access(Iss *iss, bool write, bool read) +{ + if (!iss->exec.debug_mode) + { + iss->exception.raise(iss->exec.current_insn, ISS_EXCEPT_ILLEGAL); + return false; + } + return true; +} + Cv32e40pCsr::Cv32e40pCsr(Iss &iss) : Csr(iss) { From 7944f8ed3fb569988aef9e79222d186391d97023 Mon Sep 17 00:00:00 2001 From: mpaci Date: Wed, 22 Jul 2026 06:30:13 +0200 Subject: [PATCH 15/28] fix: dret outside debug mode raises an illegal instruction The Debug spec only defines dret in debug mode, and the RTL decoder treats it as illegal elsewhere. The model executed it unconditionally and jumped through a never-written dpc. Guard dret_exec on debug_mode, matching the sfence.vma handling. --- cpu/iss_v2/include/cores/cv32e40p/priv.hpp | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp index 20f754a1..e93de9f5 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp @@ -201,6 +201,15 @@ static inline iss_reg_t mret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t dret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { + /* dret is legal only in debug mode; outside it the RTL raises an illegal + * instruction (RISC-V Debug spec, cv32e40p debug.rst). dret_handle() + * itself is unconditional (clears debug_mode, restores irq_enable, jumps + * to depc), so the guard must live here. */ + if (!iss->exec.debug_mode) + { + iss->exception.raise(pc, ISS_EXCEPT_ILLEGAL); + return pc; + } return iss->core.dret_handle(); } From b352316b544bbfbe77deec33ac81b19b4e843b95 Mon Sep 17 00:00:00 2001 From: mpaci Date: Wed, 22 Jul 2026 06:30:13 +0200 Subject: [PATCH 16/28] fix: pin dcsr.prv to M on writes (WARL on an M-only core) dcsr.prv is WARL and CV32E40P implements machine mode only, so the hardware reads back 3 regardless of what debug software writes. The view kept the written value. Mask the write down to the implemented fields (ebreakm, ebreaku, stepie, step) and force prv to M. --- cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index 1d53ed88..6b76b09c 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -361,16 +361,16 @@ bool Cv32e40pCsr::mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &va } /* RTL WARL (cv32e40p_cs_registers.sv, CSR_DCSR): writable bits are - * ebreakm(15), ebreaku(12), stepie(11), step(2) and prv[1:0] (WARL: M when - * written as M, U otherwise). xdebugver, cause and the hardwired-zero - * fields keep the stored value, which the debug entry writes directly. */ + * ebreakm(15), ebreaku(12), stepie(11) and step(2). prv[1:0] is WARL-3: + * the core is M-mode only, so any written value reads back as M. + * xdebugver, cause and the hardwired-zero fields keep the stored value, + * which the debug entry writes directly. */ bool Cv32e40pCsr::dcsr_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) { if (is_write) { constexpr iss_reg_t WRITABLE = (1u << 15) | (1u << 12) | (1u << 11) | (1u << 2); - iss_reg_t prv = ((value & 0x3) == 0x3) ? 0x3 : 0x0; - this->dcsr = (this->dcsr & ~(WRITABLE | 0x3)) | (value & WRITABLE) | prv; + this->dcsr = (this->dcsr & ~WRITABLE) | (value & WRITABLE) | 0x3; } else { From 567e6130a8f6f9e0d5a37d4739acd2e10c673aee Mon Sep 17 00:00:00 2001 From: mpaci Date: Wed, 5 Aug 2026 16:24:21 +0200 Subject: [PATCH 17/28] fix: model the CV32E40P trigger module (tdata write gating, execute match) The core implements one mcontrol trigger with an execute-address match: - tdata1 exposes bit 2 (execute match enable) as the only writable bit, tdata2 is fully writable; both writes are accepted only in debug mode and silently dropped otherwise, matching the RTL's tmatch_*_we & debug_mode_i gating (an M-mode write is NOT an illegal instruction). - Cv32e40pIrq::check() raises the debug request with cause=2 when the trigger is armed and the about-to-dispatch PC equals tdata2: the match is evaluated BEFORE execution, as in the RTL (trigger_match_o on pc_id), so the matched instruction is not retired and dpc points at it. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 1 + cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 24 +++++++++++++++++++---- cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 20 +++++++++++++++++-- 3 files changed, 39 insertions(+), 6 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index 82154fa9..19fc558a 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -182,6 +182,7 @@ class Cv32e40pCsr : public Csr bool fcsr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool hwloop_csr_access(iss_insn_t *insn, bool is_write, iss_reg_t &value, int index); bool tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool tdata_debug_gate(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool dcsr_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool dpc_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index 6b76b09c..7a12c3fb 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -197,14 +197,23 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) this->mtval.set_write_mask(0); this->mcause.set_write_mask(0x8000001F); - /* Trigger module: one trigger, tselect hardwired to 0, tdata* writable - * only from debug mode (not modelled), tinfo reports type 2. */ + /* Trigger module: one trigger, tselect hardwired to 0, tinfo reports + * type 2. tdata1 (only bit 2, execute match enable) and tdata2 (match + * address) latch only from debug mode: tmatch_control_we/tmatch_value_we + * are gated on debug_mode_i, so an M-mode write is silently dropped, + * not an illegal instruction. The execute match itself is evaluated at + * the dispatch boundary, before the matched instruction runs (see + * Cv32e40pIrq::check()). */ this->tselect.set_write_mask(0); this->tselect.register_callback(std::bind(&Cv32e40pCsr::tselect_read_zero, this, std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); this->tdata1.reset_val = 0x28001040; - this->tdata1.set_write_mask(0); - this->tdata2.set_write_mask(0); + this->tdata1.set_write_mask(0x4); + this->tdata1.register_callback(std::bind(&Cv32e40pCsr::tdata_debug_gate, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); + this->tdata2.set_write_mask(0xFFFFFFFF); + this->tdata2.register_callback(std::bind(&Cv32e40pCsr::tdata_debug_gate, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); this->tdata3.set_write_mask(0); this->declare_csr(&this->tinfo, "tinfo", 0x7A4, 0x4, 0); this->declare_csr(&this->mcontext, "mcontext", 0x7A8, 0, 0); @@ -350,6 +359,13 @@ bool Cv32e40pCsr::tselect_read_zero(iss_insn_t *insn, bool is_write, iss_reg_t & return false; } +bool Cv32e40pCsr::tdata_debug_gate(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + /* tdata1/tdata2 writes latch only in debug mode; outside it the RTL + * drops them silently (no illegal-instruction), reads are unrestricted. */ + return !is_write || this->iss.exec.debug_mode; +} + bool Cv32e40pCsr::mip_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) { /* Reads mirror the wire-driven register; writes are dropped. */ diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index b9c1132c..1a687af7 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -55,10 +55,26 @@ static int cv32e40p_irq_pick(iss_reg_t pending) int Cv32e40pIrq::check() { + /* Execute-address trigger (trigger module, mcontrol): the match fires + * BEFORE the instruction at tdata2 executes (RTL trigger_match_o on + * pc_id), entering debug with dcsr.cause=2 and dpc = the matched PC. + * Evaluated at the dispatch boundary so the matched instruction is + * never retired - a batched co-sim step cannot run past the entry. + * Only the slow dispatch handler runs check(): the co-sim personality + * pins it; a standalone fast-mode run does not evaluate triggers. */ + if ((this->iss.csr.tdata1.value & (1u << 2)) && + !this->iss.exec.debug_mode && !this->req_debug && + this->iss.exec.current_insn == this->iss.csr.tdata2.value) + { + this->req_debug = true; + this->req_debug_cause = 2; + } + /* Debug entry: generic implementation plus dcsr.cause, written * atomically with the entry as the RTL does. The cause comes from - * req_debug_cause: 3 (haltreq) on the wire path, 1 (ebreak) or 4 - * (single-step) when the bridge's informed debug entry armed it. */ + * req_debug_cause: 3 (haltreq) on the wire path, 2 (trigger) from the + * local execute-trigger match above, 1 (ebreak) or 4 (single-step) + * when armed by an external debug-entry request. */ if (this->req_debug && !this->iss.exec.debug_mode) { this->iss.exec.debug_mode = true; From 404a0305850c3ea615ca7e1ca83808da5e22fdc9 Mon Sep 17 00:00:00 2001 From: mpaci Date: Wed, 5 Aug 2026 21:55:19 +0200 Subject: [PATCH 18/28] fix: zero-latency memory ports keep the ISS at the commit boundary With latency=1 on the soc memories every load parks its sync follower in the inflight ring for one extra boundary: at any asynchronous injection point (debug entry, interrupt take) the ISS has already executed an instruction the RTL killed in ID, and no later repair can undo the writeback. latency=0 on all ports (mem, stdout, timer, debug_rom) makes every commit architecturally final at its own boundary - the contract an external lockstep driver relies on when it steps the model one commit at a time. Standalone-platform behaviour is unaffected: instruction traces carry the same retire order; only stall cycles differ. --- pulp/cv32e40p_v2_standalone.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/pulp/cv32e40p_v2_standalone.py b/pulp/cv32e40p_v2_standalone.py index 2655e03e..642e2ccc 100644 --- a/pulp/cv32e40p_v2_standalone.py +++ b/pulp/cv32e40p_v2_standalone.py @@ -120,13 +120,13 @@ def __post_init__(self): self.core = Cv32e40pConfig(isa=isa, boot_addr=self.boot_addr) # init=False: never-written bytes must read 0 (testbench memory # contract), not the 0x57 poison pattern of the default init=True. - self.mem = MemoryV3Config('mem', size=0x0040_0000, atomics=False, latency=1, + self.mem = MemoryV3Config('mem', size=0x0040_0000, atomics=False, latency=0, init=False) - self.stdout = MemoryV3Config('stdout', size=0x100, atomics=False, latency=1, + self.stdout = MemoryV3Config('stdout', size=0x100, atomics=False, latency=0, init=False) - self.timer = MemoryV3Config('timer', size=0x100, atomics=False, latency=1, + self.timer = MemoryV3Config('timer', size=0x100, atomics=False, latency=0, init=False) - self.debug_rom = MemoryV3Config('debug_rom', size=0x1000, atomics=False, latency=1, + self.debug_rom = MemoryV3Config('debug_rom', size=0x1000, atomics=False, latency=0, init=False) self.router = RouterConfig(kind='bandwidth') self.mem_mapping = RouterMapping(name='mem_mapping', From f5a8ad7c7543418eca8872f4f7ee34583b29312b Mon Sep 17 00:00:00 2001 From: mpaci Date: Thu, 6 Aug 2026 10:52:07 +0200 Subject: [PATCH 19/28] feat: native debug-entry model (ebreak, single-step, haltreq wire, informed IRQ collision) The v2 personality now models the four CV32E40P debug-entry causes natively, with the arming order in Cv32e40pIrq::check() encoding the Debug-spec priority (trigger 2 > haltreq 3 > step 4): - ebreak with dcsr.ebreakm=1 (cause 1): the new CONFIG_GVSOC_ISS_CV32E40P_V2 define (cv32e40p_v2.py) gates the hook in the shared isa/rv32i.hpp / rv32c.hpp decoders; the instruction arms the request via ebreak_enter_debug() and never retires (mcause/mepc untouched, entry row = first debug-ROM instruction, like the RTL). - single-step (cause 4): dret with dcsr.step=1 opens the window (dret_step_check, depc still live); check() re-enters once current_insn moved off step_pc - an exception redirect lands the entry on the handler address as the spec requires. Interrupts are masked inside the window unless dcsr.stepie=1. Every entry closes the window, so a haltreq during the window (or a debugger clearing dcsr.step before dret) cannot fire a stale cause=4 later. - haltreq (cause 3): first-class injector wire (RTL debug_req_i), handled model-side by haltreq_sync - it arms req_debug AND wakes a WFI-parked hart with the full three-step release, which only the model can run. The RTL sleep unit exits on debug_req_i regardless of mie/mip, while the generic release is gated on mie & mip alone: a halt request arriving with mie=0 would otherwise never wake the model. The level is tracked so a held-high line re-halts right after dret, as the level-sensitive RTL input does. - wfi_wake wire (co-simulation only, no architectural effect): pulsed externally when the DUT's retire stream proves a wake the interrupt wires cannot carry; runs the same three-step release. wfi with a pending debug request degrades to a nop (the RTL never sleeps with debug_req_i asserted). Informed interrupt+debug collision: when the DUT's entry row carries an interrupt take's CSR writes, the external lockstep driver posts the taken cause id in collide_irq_id and check() takes exactly that line ahead of the entry (irq_take, extracted from the ladder) - dpc lands on the vectored handler entry, mstatus/mepc/mcause carry the take. The model never guesses the arbitration: the outcome depends on cycle timing only the DUT observes. Injector grows from 19 to 21 lines (haltreq, wfi_wake), bound in cv32e40p_v2_standalone.py to the ports the Cv32e40pIrq constructor registers. Hardening: - check() masks the informed collision line against IRQ_MASK before irq_take (defense in depth on top of the driver-side validation) - cv32e40p_irq_pick's priority fallback would otherwise turn an unwired id into a silent phantom MTI take. - Cv32e40pIrq::reset() override clears step_state, collide_irq_id and req_debug_cause: a live single-step window or an unconsumed collision id must not survive a reset. - the wfi debugger-responsiveness guard (debug mode, dcsr.step, pending req_debug) is owned by the personality's wfi_exec - the shared IrqRiscv::wfi_handle keeps its upstream behaviour. - release_wfi() helper deduplicates the three-step release shared by haltreq_sync and wfi_wake_sync. - cv32e40p_v2.py enables the opt-in conformance the core commits gate for the other cores: CONFIG_GVSOC_ISS_RVC_STRICT and the FLEXFLOAT_TININESS_AFTER_ROUNDING c-flag. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 6 + cpu/iss_v2/include/cores/cv32e40p/irq.hpp | 85 +++++++- cpu/iss_v2/include/cores/cv32e40p/priv.hpp | 17 +- cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 190 +++++++++++++++++- pulp/cpu/iss/cv32e40p_v2.py | 18 ++ .../cv32e40p_irq_injector.cpp | 20 +- pulp/cv32e40p_v2_standalone.py | 17 +- 7 files changed, 339 insertions(+), 14 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index 19fc558a..0a36cbcf 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -112,6 +112,12 @@ class Cv32e40pCsr : public Csr * full handlers, where the event lines fire (Cv32e40pExec). */ inline bool hpm_counting(); + /* EBREAK in M-mode enters debug when dcsr.ebreakm=1 (RISC-V Debug + * Spec, dcsr bit 15). Consumed by ebreak_exec/c_ebreak_exec + * (isa/rv32i.hpp, isa/rv32c.hpp), which check debug_mode first, so + * this is only reached outside debug mode. */ + bool ebreak_m_mode_enters_debug() { return ((this->dcsr >> 15) & 1) != 0; } + /* CV32E40P-only CSRs, absent from the generic register file. */ Cv32e40pRoCsr mvendorid_ro; /* 0xF11 (replaces the base read/write reg) */ Cv32e40pRoCsr marchid_ro; /* 0xF12 (replaces the base read/write reg) */ diff --git a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp index 28e12d6d..aa1c1238 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp @@ -16,22 +16,99 @@ class Cv32e40pIrq : public IrqRiscv * sixteen fast lines irq[31:16] (cv32e40p_int_controller.sv IRQ_MASK). */ static constexpr iss_reg_t IRQ_MASK = 0xFFFF0888; - Cv32e40pIrq(Iss &iss) : IrqRiscv(iss) {} + /* Defined in irq.cpp (Iss is incomplete here): registers the haltreq + * slave port - the debug halt request line (RTL debug_req_i), a + * first-class wire like the interrupt lines, handled by haltreq_sync. */ + Cv32e40pIrq(Iss &iss); void start(); + /* Shadows IrqRiscv::reset (static dispatch via CONFIG_GVSOC_ISS_IRQ, + * iss.cpp): the base reset does not know the personality's state. A + * live single-step window (step_state/step_pc) or an unconsumed + * collision id surviving a reset would fire a phantom cause=4 entry + * or a stale interrupt take on the first post-reset boundary. */ + void reset(bool active); + /* Interrupt take with the RTL priority order and vectored entry; * shadows the generic RISC-V ladder (static dispatch via * CONFIG_GVSOC_ISS_IRQ). */ int check(); + /* haltreq wire: arms req_debug and wakes a WFI-parked hart. */ + static void haltreq_sync(vp::Block *__this, bool value); + + /* wfi_wake wire: releases a WFI-parked hart (full three-step release, + * only callable inside the model) with NO architectural side effect. + * Driven externally when the DUT's retire stream proves the wake + * happened (the RTL retires wfi at execute and sleeps after; + * wake sources like debug_req are not all visible as interrupt + * wires). */ + static void wfi_wake_sync(vp::Block *__this, bool value); + + /* ebreak with dcsr.ebreakm=1 outside debug mode (isa/rv32i.hpp, + * isa/rv32c.hpp): arms the debug request; check() performs the entry + * at the next dispatch boundary, so the ebreak never retires and + * mcause/mepc stay untouched - like the RTL, where the entry row is + * the first debug-ROM instruction. Inline: touches members only + * (Iss is incomplete here). */ + void ebreak_enter_debug() + { + this->req_debug = true; + this->req_debug_cause = 1; + } + + /* dret with dcsr.step=1 (priv.hpp dret_exec): opens the single-step + * window. check() re-enters debug with cause=4 once the stepped + * instruction is done. Defined in irq.cpp (reads dcsr/depc). */ + void dret_step_check(); + + /* Single-step window state: 0 = idle, 1 = stepping (step_pc holds the + * address of the one instruction to execute). The exit condition in + * check() is current_insn != step_pc: a completed instruction moved + * the PC, and an exception redirect (has_exception, consumed before + * check() runs) lands the entry on the handler address, as the Debug + * spec requires. A stepped jump-to-self never trips it; that corner + * is handled by an externally armed entry. */ + int step_state = 0; + iss_reg_t step_pc = 0; + + /* Level of the haltreq wire (RTL debug_req_i is level-sensitive: while + * high the hart re-halts right after dret). haltreq_sync records it; + * check() re-arms req_debug from it outside debug mode, since the wire + * only syncs on level CHANGES and a still-high level after an entry + * consumed req_debug would otherwise be lost. */ + bool haltreq_level = false; + /* dcsr.cause for the next req_debug take (debug spec: 1=ebreak, * 3=haltreq, 4=single-step). The haltreq wire path leaves the - * default; the co-simulation bridge's informed debug entry sets it - * from the DUT's dcsr before arming req_debug. Reset to 3 by the - * entry itself. */ + * default; an external debug-entry request sets it from the DUT's + * dcsr before arming req_debug. Reset to 3 by the entry itself. */ int req_debug_cause = 3; + /* Informed interrupt+debug collision (co-sim): when the DUT's debug + * entry row carries the CSR writes of an interrupt take, the + * external driver stores the taken cause id here before arming the + * entry; check() then takes exactly that line ahead of the entry, so + * dpc lands on the (vectored) handler entry and mstatus/mepc/mcause + * carry the take, as in the RTL. -1 = no collision. The model never + * guesses this on its own: the arbitration outcome depends on cycle + * timing only the DUT can observe. */ + int collide_irq_id = -1; + + vp::WireSlave haltreq_itf; + vp::WireSlave wfi_wake_itf; + private: bool mie_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value); + + /* Interrupt take (RTL priority + vectored entry); shared by the + * check() ladder and the simultaneous-interrupt debug entry. */ + void irq_take(iss_reg_t pending); + + /* Full WFI release (flag clear, retain_dec, terminate of the held + * entry - the terminated entry drains into the commit stream). Only + * callable inside the model; shared by haltreq_sync and + * wfi_wake_sync. No-op when the hart is not parked. */ + void release_wfi(); }; diff --git a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp index e93de9f5..615a6faa 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp @@ -188,7 +188,19 @@ static inline iss_reg_t csrrsi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t wfi_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - iss->irq.wfi_handle(insn); + /* wfi degrades to a nop whenever the hart must stay responsive to the + * debugger (RISC-V Debug Spec; RTL cv32e40p_sleep_unit.sv / + * controller): in debug mode, in the single-step window (dcsr.step) + * and with a pending debug request - the RTL never sleeps with + * debug_req_i asserted, and a level-high haltreq produces no fresh + * wire edge to wake a parked hart. The guard lives HERE, in the + * personality, so the shared IrqRiscv::wfi_handle keeps its + * historical behaviour for the other iss_v2 cores. */ + if (!iss->irq.req_debug && !iss->exec.debug_mode && + !((iss->csr.dcsr >> 2) & 1)) + { + iss->irq.wfi_handle(insn); + } return iss_insn_next(iss, insn, pc); } @@ -210,6 +222,9 @@ static inline iss_reg_t dret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) iss->exception.raise(pc, ISS_EXCEPT_ILLEGAL); return pc; } + /* dcsr.step=1: open the single-step window (depc still live here); + * check() re-enters debug with cause=4 after one instruction. */ + iss->irq.dret_step_check(); return iss->core.dret_handle(); } diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index 1a687af7..de2f9c0b 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -15,6 +15,20 @@ * on the inherited wire-sync path. */ #include +#include + +/* Debug halt request line (RTL debug_req_i), a first-class wire like the + * interrupt lines. Handling it inside the model lets the model run the + * full RTL semantics - in particular waking a WFI-parked hart: the RTL + * sleep unit exits on debug_req_i regardless of mie/mip, while the generic + * check_interrupts() release is gated on (mie & mip) alone. */ +Cv32e40pIrq::Cv32e40pIrq(Iss &iss) : IrqRiscv(iss) +{ + this->haltreq_itf.set_sync_meth(&Cv32e40pIrq::haltreq_sync); + this->iss.new_slave_port("haltreq", &this->haltreq_itf, (vp::Block *)this); + this->wfi_wake_itf.set_sync_meth(&Cv32e40pIrq::wfi_wake_sync); + this->iss.new_slave_port("wfi_wake", &this->wfi_wake_itf, (vp::Block *)this); +} void Cv32e40pIrq::start() { @@ -25,6 +39,94 @@ void Cv32e40pIrq::start() this, std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); } +void Cv32e40pIrq::reset(bool active) +{ + IrqRiscv::reset(active); + + if (active) + { + /* A live single-step window or an unconsumed collision id must not + * survive the reset: stale step_pc would fire a phantom cause=4 + * entry at the boot address, a stale collision id would make the + * first post-reset debug entry take an interrupt that never + * happened. haltreq_level is deliberately kept: it mirrors the + * actual wire level, which the reset does not change. */ + this->step_state = 0; + this->collide_irq_id = -1; + this->req_debug_cause = 3; + } +} + +/* Full WFI release: the same three-step sequence as check_interrupts() - + * the held WFI entry must drain into the commit stream, a bare wfi-flag + * clear would leave it parked forever. */ +void Cv32e40pIrq::release_wfi() +{ + if (this->iss.exec.wfi.get()) + { + this->iss.exec.wfi.set(false); + this->iss.exec.retain_dec(); + this->iss.exec.insn_terminate(this->wfi_entry); + } +} + +/* Debug halt request wire (RTL debug_req_i). + * + * Arms req_debug - consumed by check() at the next dispatch, exactly like + * an externally armed request - and wakes a WFI-parked hart: the + * RTL sleep unit exits sleep on debug_req_i regardless of pending + * interrupts (cv32e40p_sleep_unit.sv / controller wake-up), while the + * generic check_interrupts() release is gated on (mie & mip) alone, so a + * halt request arriving with mie=0 would otherwise never wake the model. */ +void Cv32e40pIrq::haltreq_sync(vp::Block *__this, bool value) +{ + Cv32e40pIrq *_this = (Cv32e40pIrq *)__this; + + _this->haltreq_level = value; /* check() re-arms from a held-high level */ + + if (!value) + { + return; /* level deassert: an armed req_debug stays latched */ + } + + _this->req_debug = true; /* req_debug_cause keeps its default (3 = haltreq) */ + + _this->release_wfi(); +} + +/* wfi_wake wire: releases a WFI-parked hart with no architectural side + * effect (release_wfi is only callable inside the model); driven + * externally when the DUT's retire stream proves the wake happened + * - the RTL retires wfi at execute and sleeps after, and wake sources + * like a debug_req level are not all visible as interrupt wires. The + * terminated entry drains into the commit stream, serving the DUT's own + * wfi retire. */ +void Cv32e40pIrq::wfi_wake_sync(vp::Block *__this, bool value) +{ + Cv32e40pIrq *_this = (Cv32e40pIrq *)__this; + + if (!value) + { + return; /* deassert edge of the pulse */ + } + + _this->release_wfi(); +} + +/* dret with dcsr.step=1 (priv.hpp dret_exec, called before dret_handle + * while depc is still live): opens the single-step window. The RTL + * controller re-enters debug after one instruction (cv32e40p_controller.sv + * debug_single_step_i); the matching entry is armed by check() once the + * stepped instruction is done. */ +void Cv32e40pIrq::dret_step_check() +{ + if ((this->iss.csr.dcsr >> 2) & 1) + { + this->step_state = 1; + this->step_pc = this->iss.csr.depc; + } +} + /* RTL WARL result: only the wired interrupt lines are writable in mie * (cv32e40p_cs_registers.sv, csr_mie_wdata & IRQ_MASK). */ bool Cv32e40pIrq::mie_write_fixup(iss_insn_t *insn, bool is_write, iss_reg_t &value) @@ -55,6 +157,11 @@ static int cv32e40p_irq_pick(iss_reg_t pending) int Cv32e40pIrq::check() { + /* Arming order encodes the Debug-spec cause priority (each block + * yields to an already-armed request): trigger(2) > haltreq(3) > + * step(4). The edge path (haltreq_sync) arms asynchronously and so + * outranks a same-boundary trigger match. */ + /* Execute-address trigger (trigger module, mcontrol): the match fires * BEFORE the instruction at tdata2 executes (RTL trigger_match_o on * pc_id), entering debug with dcsr.cause=2 and dpc = the matched PC. @@ -70,6 +177,32 @@ int Cv32e40pIrq::check() this->req_debug_cause = 2; } + /* Held-high haltreq re-arms (RTL debug_req_i is level-sensitive: the + * hart re-halts right after dret while the line stays asserted; the + * wire itself only syncs on level changes). */ + if (this->haltreq_level && !this->req_debug && !this->iss.exec.debug_mode) + { + this->req_debug = true; /* req_debug_cause keeps its default (3) */ + } + + /* Single-step window (dcsr.step, armed by dret_step_check): re-enter + * debug with cause=4 once the stepped instruction is done. "Done" is + * current_insn != step_pc: a completed instruction moved the PC, and + * an exception during the step lands here after the has_exception + * redirect (consumed at the top of the dispatch, before check()), so + * depc points at the handler entry, as the Debug spec requires. While + * current_insn == step_pc the instruction has not executed yet (fetch + * or scoreboard stalls re-run check() on the same boundary). The + * window is closed by the entry itself, whatever the winning cause. */ + if (this->step_state && !this->iss.exec.debug_mode && !this->req_debug) + { + if (this->iss.exec.current_insn != this->step_pc) + { + this->req_debug = true; + this->req_debug_cause = 4; + } + } + /* Debug entry: generic implementation plus dcsr.cause, written * atomically with the entry as the RTL does. The cause comes from * req_debug_cause: 3 (haltreq) on the wire path, 2 (trigger) from the @@ -77,6 +210,42 @@ int Cv32e40pIrq::check() * when armed by an external debug-entry request. */ if (this->req_debug && !this->iss.exec.debug_mode) { + /* Informed interrupt+debug collision: the RTL takes the interrupt + * first and enters debug on the first handler instruction - dpc is + * the (vectored) entry and mstatus/mepc/mcause carry the take; the + * handler instruction itself never executes. Whether the collision + * happened is decided by the DUT (its entry row carries the take's + * CSR writes), never guessed here: the arbitration outcome depends + * on cycle timing the model cannot see. Both trap_seq bumps land + * before debug_mode flips, so an external observer sees one + * atomic entry. */ + if (this->collide_irq_id >= 0) + { + /* Defense in depth: irq_take's + * priority pick falls back to MTI when no IRQ_MASK bit is set, + * so an unwired id would silently become a phantom cause-7 + * take. The id is DUT-provided (mcause & 0x1f), never trusted + * blindly. */ + iss_reg_t line = (this->collide_irq_id < 32) ? + (((iss_reg_t)1 << this->collide_irq_id) & IRQ_MASK) : 0; + if (line) + { + this->irq_take(line); + } + else + { + this->trace.msg(vp::Trace::LEVEL_WARNING, + "Informed IRQ+debug collision with unwired cause id %d - " + "take skipped\n", this->collide_irq_id); + } + this->collide_irq_id = -1; + } + + /* Any entry closes a live single-step window: without this, a + * haltreq during the window (or a dret whose debugger cleared + * dcsr.step) leaves step_state armed on a stale step_pc and the + * next boundary would fire a spurious cause=4 entry. */ + this->step_state = 0; this->iss.exec.debug_mode = true; this->iss.csr.depc = this->iss.exec.current_insn; this->iss.csr.dcsr = (this->iss.csr.dcsr & ~(0x7u << 6)) | @@ -101,6 +270,14 @@ int Cv32e40pIrq::check() return 0; } + /* Inside the single-step window interrupts are masked unless + * dcsr.stepie=1 (bit 11): the RTL controller holds irq_req off while + * single-stepping (cv32e40p_controller.sv debug_single_step_i). */ + if (this->step_state && !((this->iss.csr.dcsr >> 11) & 1)) + { + return 0; + } + /* M-mode only core: the take needs a wired pending line and the global * enable (mstatus.MIE). */ iss_reg_t pending = this->iss.csr.mie.value & this->iss.csr.mip.value & IRQ_MASK; @@ -109,6 +286,17 @@ int Cv32e40pIrq::check() return 0; } + this->irq_take(pending); + + return 1; +} + +/* Interrupt take with the RTL priority order and vectored entry. The + * caller guarantees at least one bit of pending (mie & mip & IRQ_MASK) + * and mstatus.MIE=1. Shared by the ladder in check() and the + * simultaneous-interrupt path of the debug entry. */ +void Cv32e40pIrq::irq_take(iss_reg_t pending) +{ int irq = cv32e40p_irq_pick(pending); /* mtvec holds {base[31:8], 0, mode} (Cv32e40pCsr::mtvec_write_fixup); @@ -133,6 +321,4 @@ int Cv32e40pIrq::check() this->irq_enable.set(0); this->iss.timing.stall_insn_dependency_account(4); - - return 1; } diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index f9597703..0fd13123 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -99,6 +99,24 @@ class Cv32e40pIrq(IssModule): def gen(self, iss: RiscvCommon): iss.isa.add_define('CONFIG_GVSOC_ISS_IRQ', 'Cv32e40pIrq') iss.isa.add_define('CONFIG_GVSOC_ISS_RISCV_EXCEPTIONS', 1) + # Marks the CV32E40P iss_v2 personality build for the shared ISA + # headers: gates the debug-entry hooks (ebreak with dcsr.ebreakm, + # dret single-step window) in isa/rv32i.hpp and isa/rv32c.hpp. + # The v1 flag CONFIG_GVSOC_ISS_CV32E40P must stay off here (it + # also gates v1-only core.hpp/csr.hpp/dbgunit paths). + iss.isa.add_define('CONFIG_GVSOC_ISS_CV32E40P_V2', 1) + # Strict RVC decoding (isa/rv32c.hpp): reserved code-points + # (c.addi4spn nzuimm=0, c.addi16sp/c.lui imm=0, c.lwsp rd=0, + # c.jr rs1=0) raise illegal-instruction, as the CV32E40P RTL does. + # Opt-in so the other cores keep the historical permissive + # decoding (and their golden traces) by default. + iss.isa.add_define('CONFIG_GVSOC_ISS_RVC_STRICT', 1) + # IEEE754 leaves the tininess detection point to the + # implementation: FPnew (the CV32E40P FPU) detects it after + # rounding with unbounded exponent (fpnew_fma.sv:616). Opt-in for + # parity with the RTL; other cores keep flexfloat's default + # (before-rounding) so their FP flags are unchanged. + iss.add_c_flags(['-DFLEXFLOAT_TININESS_AFTER_ROUNDING=1']) iss.isa.add_include('') iss.add_sources([ 'cpu/iss_v2/src/irq/irq_riscv.cpp', diff --git a/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp b/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp index 67f6e51a..f138a52c 100644 --- a/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp +++ b/pulp/cv32e40p_irq_injector/cv32e40p_irq_injector.cpp @@ -21,9 +21,9 @@ /* * CV32E40P interrupt-line injector. * - * Exposes the core interrupt wires to an external gv:: client: the - * co-simulation bridge binds each line with gv::wire_bind and drives it - * as the RTL irq_i inputs change. Every line forwards to the matching + * Exposes the core interrupt wires to an external gv:: client, which + * binds each line with gv::wire_bind and drives it as the RTL irq_i + * inputs change. Every line forwards to the matching * IrqRiscv slave port, so mip and the wake-up logic follow the same path * as a platform interrupt source. */ @@ -59,8 +59,14 @@ class Cv32e40pIrqInjector : public vp::Component vp::WireMaster itf; }; - /* msi, mti, mei plus the sixteen fast lines irq[31:16]. */ - static constexpr int NB_LINES = 19; + /* msi, mti, mei, the sixteen fast lines irq[31:16], plus the debug + * halt request (RTL debug_req_i - wakes a WFI-parked hart, so it must + * travel the wire path like the interrupt lines, not a struct write) + * and the wfi_wake release (externally driven, no architectural + * effect: fires when the DUT's stream proves a wake the wires cannot + * carry). */ + static constexpr int NB_IRQ_LINES = 19; + static constexpr int NB_LINES = 21; Line lines[NB_LINES]; }; @@ -70,10 +76,12 @@ Cv32e40pIrqInjector::Cv32e40pIrqInjector(vp::ComponentConf &config) this->lines[0].name = "msi"; this->lines[1].name = "mti"; this->lines[2].name = "mei"; - for (int i = 3; i < NB_LINES; i++) + for (int i = 3; i < NB_IRQ_LINES; i++) { this->lines[i].name = "external_irq_" + std::to_string(16 + i - 3); } + this->lines[NB_IRQ_LINES].name = "haltreq"; + this->lines[NB_IRQ_LINES + 1].name = "wfi_wake"; for (auto &line : this->lines) { diff --git a/pulp/cv32e40p_v2_standalone.py b/pulp/cv32e40p_v2_standalone.py index 642e2ccc..ae7fae23 100644 --- a/pulp/cv32e40p_v2_standalone.py +++ b/pulp/cv32e40p_v2_standalone.py @@ -180,12 +180,27 @@ def __init__(self, parent, name, config: Cv32e40pStandaloneConfig, binary): core.o_FETCH ( ico.i_INPUT(1) ) core.o_DATA ( ico.i_INPUT(2) ) - # Interrupt lines: the co-sim bridge drives the injector through + # Interrupt lines: an external client drives the injector through # gv::wire_bind; each line lands on the core's native slave port, # so mip and the WFI wake-up follow the hardware path. irq_inj = Cv32e40pIrqInjector(self, 'irq_injector') for name, irq in zip(IRQ_LINES, IRQ_NUMBERS): irq_inj.o_LINE(name, core.i_IRQ(irq)) + # Debug halt request (RTL debug_req_i): same wire path as the + # interrupt lines. Handled by Cv32e40pIrq::haltreq_sync, which arms + # req_debug and wakes a WFI-parked hart (the RTL sleep unit exits on + # debug_req_i regardless of mie/mip). The port is CV32E40P-specific + # (registered by the Cv32e40pIrq constructor), hence the inline + # SlaveItf instead of a riscv.py getter. + irq_inj.o_LINE('haltreq', gvsoc.systree.SlaveItf( + core, itf_name='haltreq', signature='wire')) + # WFI release (co-simulation only): pulsed externally when the + # DUT's retire stream proves the hart woke (the RTL retires wfi at + # execute and sleeps after; wakes like a debug_req level are not + # all visible as interrupt wires). Cv32e40pIrq::wfi_wake_sync runs + # the three-step release with no architectural side effect. + irq_inj.o_LINE('wfi_wake', gvsoc.systree.SlaveItf( + core, itf_name='wfi_wake', signature='wire')) class Cv32e40pStandaloneTop(gvsoc.systree.Component): From 9d6d63618328edfa008e7331198d79feee02e520 Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 7 Aug 2026 10:29:49 +0200 Subject: [PATCH 20/28] feat: certify externally armed IRQ collide takes at the dispatch boundary An IRQ take adjacent to a debug entry lands its mcause on the row BEFORE the entry row; an external lockstep driver can only nominate a candidate (wired id + the take's expected mepc), since only the model knows the dispatch boundary. The model certifies the candidate in Cv32e40pIrq::check(), where current_insn IS the boundary: the take fires only when the boundary matches the expected mepc, stale candidates are discarded with a warning. New fields collide_expected_mepc / collide_certify, armed alongside collide_irq_id and cleared on every check()/reset(). --- cpu/iss_v2/include/cores/cv32e40p/irq.hpp | 10 ++++++++++ cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 21 ++++++++++++++++++++- 2 files changed, 30 insertions(+), 1 deletion(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp index aa1c1238..9962ec55 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp @@ -96,6 +96,16 @@ class Cv32e40pIrq : public IrqRiscv * timing only the DUT can observe. */ int collide_irq_id = -1; + /* Certification of an adjacent-row collision candidate (the take's + * mcause rode the row BEFORE the entry row): the take fires only if + * the entry boundary (current_insn, the future depc source) equals + * the take's mepc - a stale-mcause candidate fails this by + * construction. Checked here, at the entry itself: only the model + * knows the boundary at dispatch time. collide_certify=false keeps + * the unconditional same-row behaviour. */ + iss_reg_t collide_expected_mepc = 0; + bool collide_certify = false; + vp::WireSlave haltreq_itf; vp::WireSlave wfi_wake_itf; diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index de2f9c0b..dc5accf0 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -53,6 +53,7 @@ void Cv32e40pIrq::reset(bool active) * actual wire level, which the reset does not change. */ this->step_state = 0; this->collide_irq_id = -1; + this->collide_certify = false; this->req_debug_cause = 3; } } @@ -228,7 +229,24 @@ int Cv32e40pIrq::check() * blindly. */ iss_reg_t line = (this->collide_irq_id < 32) ? (((iss_reg_t)1 << this->collide_irq_id) & IRQ_MASK) : 0; - if (line) + /* Adjacent-row candidates carry the take's mepc and are + * certified HERE, where current_insn is the entry boundary + * (the future depc source): a stale-mcause candidate - a take + * this hart already followed rows ago - parks the boundary + * elsewhere and is discarded without a take. Same-row + * candidates (collide_certify=false) keep the unconditional + * behaviour. */ + if (this->collide_certify && + this->iss.exec.current_insn != this->collide_expected_mepc) + { + this->trace.msg(vp::Trace::LEVEL_WARNING, + "Informed IRQ+debug collision id %d discarded: entry " + "boundary 0x%x != take mepc 0x%x\n", + this->collide_irq_id, + (unsigned)this->iss.exec.current_insn, + (unsigned)this->collide_expected_mepc); + } + else if (line) { this->irq_take(line); } @@ -239,6 +257,7 @@ int Cv32e40pIrq::check() "take skipped\n", this->collide_irq_id); } this->collide_irq_id = -1; + this->collide_certify = false; } /* Any entry closes a live single-step window: without this, a From 4158d1387c53249cf8d0da33f4e64cb16990afd2 Mon Sep 17 00:00:00 2001 From: mpaci Date: Fri, 7 Aug 2026 10:30:02 +0200 Subject: [PATCH 21/28] feat: stamp each commit-stream entry with the trapped bit The external stepper needs to know, on rvfi_trap rows, whether the committed instruction itself raised an architectural exception: an ecall/ebreak commit is the faulting step and must be consumed there, while a pipeline kill-and-replay row commits a NORMAL instruction the DUT re-executes on the next row - consuming it there shifts the compare stream by one retire. The trap_seq stamp cannot separate the two (the faulting insn is stamped after its own raise), so retire_account records exec.has_exception into commit_trapped[], carried through the inflight ring at drain like the trap_seq stamp. --- cpu/iss_v2/include/cores/cv32e40p/events.hpp | 11 +++++++++++ cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp | 6 ++++++ 2 files changed, 17 insertions(+) diff --git a/cpu/iss_v2/include/cores/cv32e40p/events.hpp b/cpu/iss_v2/include/cores/cv32e40p/events.hpp index fdd825f6..2506953a 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events.hpp @@ -56,6 +56,16 @@ class Cv32e40pEvents : public Events uint64_t trap_seq = 0; uint64_t commit_trap_seq[COMMIT_RING]; + /* Whether the committed instruction itself TRAPPED (exec.has_exception + * at the retire hook). The external stepper needs this on trap rows: + * an ecall/ebreak commit is the faulting step and must be consumed + * there, while a pipeline kill-and-replay row (rvfi_trap with no + * architectural trap) commits a NORMAL instruction that the DUT + * re-executes on the next row - consuming it there would shift the + * compare stream by one retire. The trap_seq stamp cannot separate + * the two: the faulting insn is stamped after its own raise(). */ + bool commit_trapped[COMMIT_RING]; + /* Drop commits not consumed yet (external resync forced a new PC). */ inline void commit_stream_flush(); @@ -73,6 +83,7 @@ class Cv32e40pEvents : public Events uint64_t inflight_pop = 0; iss_reg_t inflight_pc[COMMIT_RING]; uint64_t inflight_trap_seq[COMMIT_RING]; + bool inflight_trapped[COMMIT_RING]; /* Event lines fired by the executing instruction, committed as one OR * mask at retire: each counter advances at most +1 per instruction, as diff --git a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp index 64fde14b..f7b6f267 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp @@ -54,6 +54,8 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) * this same program order. */ this->inflight_pc[this->inflight_push % COMMIT_RING] = insn->addr; this->inflight_trap_seq[this->inflight_push % COMMIT_RING] = this->trap_seq; + this->inflight_trapped[this->inflight_push % COMMIT_RING] = + this->iss.exec.has_exception; this->inflight_push++; } else @@ -61,6 +63,8 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) { this->commit_pc[this->commit_push % COMMIT_RING] = insn->addr; this->commit_trap_seq[this->commit_push % COMMIT_RING] = this->trap_seq; + this->commit_trapped[this->commit_push % COMMIT_RING] = + this->iss.exec.has_exception; this->commit_push++; } /* A trapping instruction does not retire: drop its event lines. */ @@ -86,6 +90,8 @@ inline void Cv32e40pEvents::insn_stall_account() this->inflight_pc[this->inflight_pop % COMMIT_RING]; this->commit_trap_seq[this->commit_push % COMMIT_RING] = this->inflight_trap_seq[this->inflight_pop % COMMIT_RING]; + this->commit_trapped[this->commit_push % COMMIT_RING] = + this->inflight_trapped[this->inflight_pop % COMMIT_RING]; this->inflight_pop++; this->commit_push++; } From f5c17072178c493f5e45ad7bd79da7a88350658e Mon Sep 17 00:00:00 2001 From: mpaci Date: Tue, 11 Aug 2026 02:15:21 +0200 Subject: [PATCH 22/28] feat(cv32e40p): route debug-mode exceptions and mret to dm_exception_addr CV32E40P UM (debug.rst): an exception taken in debug mode jumps to dm_exception_addr without updating mepc/mcause/mstatus or the privilege mode, and mret/uret in debug mode jump back to this address "without affecting status registers". The generic paths went to mtvec (clobbering the CSRs) and to mepc respectively. Cv32e40pException::raise gains the debug-mode branch and reads the new debug_exception_handler config key (falls back to the debug handler entry when absent); Cv32e40pCore::mret_handle redirects before the generic side effects. The platform passes the testbench default 0x1A111600 (uvme_cv32e40p_cfg). Exercised by debug_test tests 11/12/13 (illegal CSR access, ecall and mret inside the debugger). --- .../include/cores/cv32e40p/exception.hpp | 6 ++++- cpu/iss_v2/src/cores/cv32e40p/core.cpp | 11 +++++++++ cpu/iss_v2/src/cores/cv32e40p/exception.cpp | 24 +++++++++++++++++++ pulp/cpu/iss/cv32e40p_v2.py | 9 ++++--- 4 files changed, 46 insertions(+), 4 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/exception.hpp b/cpu/iss_v2/include/cores/cv32e40p/exception.hpp index 8def012e..150f54f5 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/exception.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/exception.hpp @@ -12,12 +12,16 @@ class Cv32e40pException : public Exception { public: - Cv32e40pException(Iss &iss) : Exception(iss), iss(iss) {} + /* Defined in the .cpp: reads the debug_exception_handler config key, + * which needs the complete Iss type. */ + Cv32e40pException(Iss &iss); /* Exception entry at the mtvec base; shadows the generic raise * (static dispatch via CONFIG_GVSOC_ISS_EXCEPTION). */ void raise(iss_reg_t pc, int id); + iss_addr_t debug_exception_handler_addr; + private: Iss &iss; }; diff --git a/cpu/iss_v2/src/cores/cv32e40p/core.cpp b/cpu/iss_v2/src/cores/cv32e40p/core.cpp index 4f2c7a03..f36fd863 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/core.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/core.cpp @@ -8,6 +8,17 @@ iss_reg_t Cv32e40pCore::mret_handle() { + /* MRET executed while in debug mode (CV32E40P UM, debug.rst): the PC + * jumps to dm_exception_addr "without affecting status registers" - + * no mstatus/mcause/privilege side effects, the hart stays in debug + * mode. The generic handler below would return to mepc and restore + * mstatus.mie. */ + if (this->iss.exec.debug_mode) + { + this->iss.exec.switch_to_full_mode(); + return this->iss.exception.debug_exception_handler_addr & ~(iss_reg_t)0x3; + } + /* mcause holds its value across MRET on CV32E40P (cleared only by the * next trap or an explicit CSR write); the generic handler zeroes it. */ iss_reg_t mcause = this->iss.csr.mcause.value; diff --git a/cpu/iss_v2/src/cores/cv32e40p/exception.cpp b/cpu/iss_v2/src/cores/cv32e40p/exception.cpp index 18ac5c2a..f296bf6b 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/exception.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/exception.cpp @@ -6,11 +6,35 @@ #include +Cv32e40pException::Cv32e40pException(Iss &iss) +: Exception(iss), iss(iss) +{ + /* dm_exception_addr_i of the RTL: exceptions taken while in debug + * mode enter here (RISC-V debug spec 0.13.2). Falls back to + * the debug handler entry when the platform config lacks the key. */ + js::Config *conf = iss.get_js_config()->get("debug_exception_handler"); + this->debug_exception_handler_addr = + conf != NULL ? (iss_addr_t)conf->get_int() : this->debug_handler_addr; +} + void Cv32e40pException::raise(iss_reg_t pc, int id) { /* Commit-stream consumers gate state compares on this (events.hpp). */ this->iss.timing.trap_seq++; + /* Exception taken while in debug mode (RISC-V debug spec 0.13.2, + * CV32E40P manual): the hart jumps to dm_exception_addr and stays in + * debug mode; mepc/mcause/mstatus and the privilege mode are NOT + * updated. The generic raise would route to mtvec and clobber them. */ + if (id != ISS_EXCEPT_DEBUG && this->iss.exec.debug_mode) + { + this->iss.exec.switch_to_full_mode(); + this->iss.exec.has_exception = true; + this->iss.exec.exception_pc = + this->debug_exception_handler_addr & ~(iss_reg_t)0x3; + return; + } + this->Exception::raise(pc, id); /* Exceptions enter at the mtvec base. mtvec.value keeps the RTL mode diff --git a/pulp/cpu/iss/cv32e40p_v2.py b/pulp/cpu/iss/cv32e40p_v2.py index 0fd13123..94d9a3c9 100644 --- a/pulp/cpu/iss/cv32e40p_v2.py +++ b/pulp/cpu/iss/cv32e40p_v2.py @@ -302,8 +302,11 @@ def __init__(self, parent: Component, name: str, config: Cv32e40pConfig, 'hwloop': Hwloop(), } - # dm_halt_addr of the RTL testbench: the debug entry redirects here - # (linker script `dbg` region, loaded from the test ELF). + # dm_halt_addr / dm_exception_addr of the RTL testbench: the debug + # entry redirects to the first, exceptions taken while in debug mode + # to the second (linker script `dbg` region, loaded from the test + # ELF; uvme_cv32e40p_cfg defaults). super().__init__(parent, name, config=config, isa=isa_instance, misa=misa, zfinx=zfinx, modules=modules, - debug_handler=0x1A110800) + debug_handler=0x1A110800, + debug_exception_handler=0x1A111600) From 5521cebd4dbe9d26bb6649dfd52ae3940145321c Mon Sep 17 00:00:00 2001 From: mpaci Date: Tue, 11 Aug 2026 02:15:21 +0200 Subject: [PATCH 23/28] fix(cv32e40p): arbitrate a trigger match against armed debug requests like the RTL Two entry paths, two rules (cv32e40p_controller.sv): on a shared DBG_TAKEN_ID boundary the trigger outranks a haltreq that latched asynchronously before it (priority table, trigger highest), while a closing single-step window enters through DBG_TAKEN_IF, whose cause mux never looks at trigger_match - step wins and the trigger fires on the following session. The trigger arm now overrides an armed cause-3 request but yields when the step window closes on the boundary; a window still sitting on its own step_pc (dret straight onto the matched address) keeps the trigger-first behaviour. The trigger match - synchronous debug, not an asynchronous event - is evaluated ahead of the internal async gate (skip_irq_check, consumed inside check() core-side) on EVERY dispatch boundary and falls through to the entry when it arms; the haltreq re-arm, step window and interrupt ladder stay behind the gate. Both directions were wrong before: dcsr.cause read 3 where the DUT said 2 (debug_test_trigger) and 2 where the DUT said 4 (debug_test, single-step region). --- cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 47 +++++++++++++++++++++++---- 1 file changed, 40 insertions(+), 7 deletions(-) diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index dc5accf0..bcb5ccda 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -158,10 +158,17 @@ static int cv32e40p_irq_pick(iss_reg_t pending) int Cv32e40pIrq::check() { - /* Arming order encodes the Debug-spec cause priority (each block - * yields to an already-armed request): trigger(2) > haltreq(3) > - * step(4). The edge path (haltreq_sync) arms asynchronously and so - * outranks a same-boundary trigger match. */ + /* Cause arbitration on a shared entry boundary follows the RTL, which + * has TWO distinct entry paths: + * - DBG_TAKEN_ID (kill of the ID insn): TRIGGER (highest) > EBREAK > + * HALTREQ. The trigger block therefore OVERRIDES an armed haltreq - + * e.g. the async wire edge that latched before the matched boundary; + * the level re-arm serves it after dret, like the RTL re-halt. + * - DBG_TAKEN_IF (single-step window close): its cause mux never + * looks at trigger_match, so an armed STEP request (cause 4) is + * NOT overridden even when the next insn sits at tdata2 - the + * trigger fires on the following session instead. An armed cause + * 1/2 (injected ebreak/trigger, DUT-observed) is left alone too. */ /* Execute-address trigger (trigger module, mcontrol): the match fires * BEFORE the instruction at tdata2 executes (RTL trigger_match_o on @@ -170,14 +177,40 @@ int Cv32e40pIrq::check() * never retired - a batched co-sim step cannot run past the entry. * Only the slow dispatch handler runs check(): the co-sim personality * pins it; a standalone fast-mode run does not evaluate triggers. */ - if ((this->iss.csr.tdata1.value & (1u << 2)) && - !this->iss.exec.debug_mode && !this->req_debug && - this->iss.exec.current_insn == this->iss.csr.tdata2.value) + /* The step-window guard mirrors DBG_TAKEN_IF: when the window closes + * at this boundary (armed and the stepped insn is done) the step entry + * wins even if the NEXT insn sits at tdata2 - the trigger block runs + * first in program order, so it must yield explicitly. A window still + * on its own step_pc (dret straight onto the matched insn) does not + * close here and the trigger fires as on the RTL. */ + bool trigger_match = + (this->iss.csr.tdata1.value & (1u << 2)) && + !this->iss.exec.debug_mode && + (!this->req_debug || this->req_debug_cause == 3) && + !(this->step_state && this->iss.exec.current_insn != this->step_pc) && + this->iss.exec.current_insn == this->iss.csr.tdata2.value; + if (trigger_match) { this->req_debug = true; this->req_debug_cause = 2; } + /* One-shot async gate, consumed AFTER the trigger match: the execute + * trigger is synchronous debug (mcontrol timing=before), not an + * asynchronous event, so it fires even on a suppressed dispatch and + * falls through to the entry below. Everything past this point - + * haltreq wire re-arm, step window, interrupt ladder - is asynchronous + * and stays out of a suppressed boundary (DPI lockstep stepping: the + * engine holds the line high and injects takes explicitly). */ + if (this->iss.exec.skip_irq_check) + { + this->iss.exec.skip_irq_check = false; + if (!trigger_match) + { + return 0; + } + } + /* Held-high haltreq re-arms (RTL debug_req_i is level-sensitive: the * hart re-halts right after dret while the line stays asserted; the * wire itself only syncs on level changes). */ From a5cd7f4ed9386859c8cbe54fdd5dfdfcc120fa69 Mon Sep 17 00:00:00 2001 From: mpaci Date: Wed, 12 Aug 2026 16:55:56 +0200 Subject: [PATCH 24/28] fix(cv32e40p): the PMP CSR bank must raise illegal instruction CV32E40P has no PMP: the UM's CSR chapter lists no pmpcfg/pmpaddr bank and the RTL raises illegal instruction on any access. The generic model declares the whole bank (pmpcfg0..15, pmpaddr0..63) even when the PMP module is the empty variant - CONFIG_GVSOC_ISS_PMP is a type name and is always defined - so a read returned zero instead of trapping. Undeclare the bank alongside the other nonexistent CSRs; the raise-on-unsupported path then matches the hardware. Found by the all_csr_por CSR sweep: it diverged at pmpcfg0 after 1.16M clean retires (the DUT entered the illegal handler, the ISS carried on into pmpcfg1). With the fix the full sweep passes cleanly; cv32e40p_csr_access_test, modeled_csr_por and readonly_csr_access pass unchanged. --- cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index 7a12c3fb..4ca5e907 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -90,6 +90,19 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) this->undeclare_csr(addr); } + /* No PMP: the UM's CSR chapter has no pmpcfg/pmpaddr bank and the RTL + * raises illegal on any access. The generic model declares the whole + * bank even when the PMP module is the empty variant + * (CONFIG_GVSOC_ISS_PMP is a type name, always defined). */ + for (iss_reg_t addr = 0x3A0; addr < 0x3B0; addr++) /* pmpcfg0..15 */ + { + this->undeclare_csr(addr); + } + for (iss_reg_t addr = 0x3B0; addr < 0x3F0; addr++) /* pmpaddr0..63 */ + { + this->undeclare_csr(addr); + } + this->raise_on_unsupported_csr = true; /* Machine information registers: read-only, writes raise illegal. From 9d65a2cf43cb4af9bc1f847af37417d60b36d363 Mon Sep 17 00:00:00 2001 From: mpaci Date: Mon, 17 Aug 2026 01:54:08 +0200 Subject: [PATCH 25/28] fix(cv32e40p): hold async takes for the co-sim driver; carry insn and count retires per spec Two extensions to the model's external-stepper contract, both inert in standalone runs. Async-event hold (Cv32e40pIrq::dpi_async_hold). exec.skip_irq_check is a one-shot consumed at the first check() of a step quantum, but a 20 ns engine quantum runs several dispatches: every dispatch after the first ran unguarded, and the model took pending wire IRQs (or re-armed haltreq) at a boundary of its own stepping cadence, racing the DUT's entry row. The hold is a LEVEL the driver owns: asynchronous events are delivered as state (mip from the wires, req_debug latched) but never taken at a model-chosen boundary; the take_irq / take_debug injection windows lower it so the entry lands on the DUT-proven boundary. Synchronous conditions - execute-address trigger, single-step window close, ebreak-to-debug, driver-armed causes - keep their architectural timing and ignore the hold. Retire accounting (Cv32e40pEvents / Cv32e40pCsr). The commit ring now carries the raw encoding next to the PC (truncated to 16 bits on RVC rows: the fetched word holds the next parcel in its upper half), through the inflight FIFO as well, so the stepper can serve a live instruction-binary compare. minstret follows the RTL event exactly (cv32e40p_id_stage.sv:1639 and :1659): EBREAK never counts - the trapping forms were already dropped with the event lines, count_instr covers the debug-entry ebreak that retires without a trap - and a CSR write to either counter half suppresses that row's increment (cs_registers write-wins gate), armed by the new write callbacks and consumed at the same retire in hpm_commit. Validated in step-and-compare with minstret/minstreth/instreth and the instruction-binary compare enabled. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 28 ++++++++++++---- cpu/iss_v2/include/cores/cv32e40p/events.hpp | 6 ++++ .../include/cores/cv32e40p/events_implem.hpp | 23 +++++++++++-- cpu/iss_v2/include/cores/cv32e40p/irq.hpp | 25 ++++++++++++++ cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 33 +++++++++++++++++++ cpu/iss_v2/src/cores/cv32e40p/irq.cpp | 29 ++++++++++++++-- 6 files changed, 133 insertions(+), 11 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index 0a36cbcf..d1b98038 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -104,9 +104,11 @@ class Cv32e40pCsr : public Csr void fp_state_dirty(); /* Advance the counters for one retired instruction: events is the OR of - * the RTL hpm_events lines it fired (see cores/cv32e40p/events.hpp). - * Called once per retire by Cv32e40pEvents::event_retire_account. */ - inline void hpm_commit(uint32_t events); + * the RTL hpm_events lines it fired (see cores/cv32e40p/events.hpp); + * count_instr is the RTL minstret event line (false for EBREAK, which + * never counts - cv32e40p_id_stage.sv:1639). Called once per retire by + * Cv32e40pEvents::event_retire_account. */ + inline void hpm_commit(uint32_t events, bool count_instr); /* True while any implemented counter is enabled: keeps the core on the * full handlers, where the event lines fire (Cv32e40pExec). */ @@ -196,6 +198,8 @@ class Cv32e40pCsr : public Csr bool dscratch1_view_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool minstret_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); + bool minstreth_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool cycle_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool cycleh_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); bool instret_alias_access(iss_insn_t *insn, bool is_write, iss_reg_t &value); @@ -212,6 +216,13 @@ class Cv32e40pCsr : public Csr void mcycle_set(uint64_t count); int64_t mcycle_offset = 0; + + /* Set by a CSR write to minstret/minstreth, consumed (and cleared) by + * hpm_commit at that same instruction's retire: the RTL suppresses the + * minstret increment on the cycle the counter is written + * (cv32e40p_cs_registers.sv, !write_lower && !write_upper gate), so + * the csrw itself must not count on top of the written value. */ + bool minstret_written = false; }; inline bool Cv32e40pCsr::fp_access_illegal() @@ -236,10 +247,15 @@ inline bool Cv32e40pCsr::hpm_counting() return (this->mcountinhibit.value & event_bits) != event_bits; } -inline void Cv32e40pCsr::hpm_commit(uint32_t events) +inline void Cv32e40pCsr::hpm_commit(uint32_t events, bool count_instr) { - /* minstret: retired instructions, gated on mcountinhibit.IR (bit 2). */ - if (!(this->mcountinhibit.value & 0x4)) + /* minstret: retired instructions, gated on mcountinhibit.IR (bit 2), + * on the RTL event line (count_instr, false for EBREAK) and on the + * same-row write suppression. The flag clears unconditionally: it + * belongs to this retire only. */ + bool wrote_counter = this->minstret_written; + this->minstret_written = false; + if (count_instr && !wrote_counter && !(this->mcountinhibit.value & 0x4)) { if (++this->minstret.value == 0) { diff --git a/cpu/iss_v2/include/cores/cv32e40p/events.hpp b/cpu/iss_v2/include/cores/cv32e40p/events.hpp index 2506953a..c9ffacfd 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events.hpp @@ -47,6 +47,11 @@ class Cv32e40pEvents : public Events uint64_t commit_pop = 0; iss_reg_t commit_pc[COMMIT_RING]; + /* Raw encoding of the committed instruction (insn->opcode at the push + * site). The external stepper forwards it for the RVVI INSBIN compare + * against the DUT's rvfi_insn. Same push/pop discipline as commit_pc. */ + iss_reg_t commit_insn[COMMIT_RING]; + /* Trap-redirect sequence, bumped by Cv32e40pException::raise and the * Cv32e40pIrq take. Each commit entry is stamped with the value seen * when the instruction executed; a stamp older than the current @@ -82,6 +87,7 @@ class Cv32e40pEvents : public Events uint64_t inflight_push = 0; uint64_t inflight_pop = 0; iss_reg_t inflight_pc[COMMIT_RING]; + iss_reg_t inflight_insn[COMMIT_RING]; uint64_t inflight_trap_seq[COMMIT_RING]; bool inflight_trapped[COMMIT_RING]; diff --git a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp index f7b6f267..edf2537d 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/events_implem.hpp @@ -46,6 +46,10 @@ inline void Cv32e40pEvents::event_jump_account() inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) { Events::event_retire_account(insn); + /* Encoding for the RVVI INS compare. The fetched word carries the NEXT + * parcel in its upper half on RVC rows, so truncate to the insn size + * (the bridge masks the DUT side the same way). */ + iss_reg_t enc = (insn->size == 2) ? (insn->opcode & 0xFFFF) : insn->opcode; #ifdef CONFIG_GVSOC_ISS_EXEC_INORDER_COMMIT if (this->iss.exec.queue_head != NULL) { @@ -53,6 +57,7 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) * head): visible at drain time, through insn_stall_account, in * this same program order. */ this->inflight_pc[this->inflight_push % COMMIT_RING] = insn->addr; + this->inflight_insn[this->inflight_push % COMMIT_RING] = enc; this->inflight_trap_seq[this->inflight_push % COMMIT_RING] = this->trap_seq; this->inflight_trapped[this->inflight_push % COMMIT_RING] = this->iss.exec.has_exception; @@ -62,6 +67,7 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) #endif { this->commit_pc[this->commit_push % COMMIT_RING] = insn->addr; + this->commit_insn[this->commit_push % COMMIT_RING] = enc; this->commit_trap_seq[this->commit_push % COMMIT_RING] = this->trap_seq; this->commit_trapped[this->commit_push % COMMIT_RING] = this->iss.exec.has_exception; @@ -73,10 +79,19 @@ inline void Cv32e40pEvents::event_retire_account(iss_insn_t *insn) this->pending_events = 0; return; } - uint32_t events = this->pending_events | CV32E40P_HPM_INSTR - | (insn->size == 2 ? CV32E40P_HPM_COMP_INSTR : 0); + /* RTL minstret event (cv32e40p_id_stage.sv:1639) excludes EBREAK + * unconditionally; the compressed-retired event shares the gate + * (:1664, minstret && is_compressed). The trapping ebreak forms were + * dropped above with has_exception - this covers the debug-entry + * ebreak (dcsr.ebreakm=1), which retires without an architectural + * trap yet must not count. */ + bool count_instr = !(enc == 0x00100073u + || (insn->size == 2 && enc == 0x9002u)); + uint32_t events = this->pending_events + | (count_instr ? (CV32E40P_HPM_INSTR + | (insn->size == 2 ? CV32E40P_HPM_COMP_INSTR : 0)) : 0); this->pending_events = 0; - this->iss.csr.hpm_commit(events); + this->iss.csr.hpm_commit(events, count_instr); } inline void Cv32e40pEvents::insn_stall_account() @@ -88,6 +103,8 @@ inline void Cv32e40pEvents::insn_stall_account() { this->commit_pc[this->commit_push % COMMIT_RING] = this->inflight_pc[this->inflight_pop % COMMIT_RING]; + this->commit_insn[this->commit_push % COMMIT_RING] = + this->inflight_insn[this->inflight_pop % COMMIT_RING]; this->commit_trap_seq[this->commit_push % COMMIT_RING] = this->inflight_trap_seq[this->inflight_pop % COMMIT_RING]; this->commit_trapped[this->commit_push % COMMIT_RING] = diff --git a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp index 9962ec55..10d9d123 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/irq.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/irq.hpp @@ -106,6 +106,31 @@ class Cv32e40pIrq : public IrqRiscv iss_reg_t collide_expected_mepc = 0; bool collide_certify = false; + /* Co-sim asynchronous-event hold (level, external-driver owned). + * + * While set, check() delivers asynchronous external events as STATE + * (mip via the wires, req_debug latched from the haltreq level) but + * never TAKES them at a dispatch boundary: no interrupt-ladder take, + * no haltreq re-arm, no cause-3 (haltreq) debug entry. The take + * boundary of an asynchronous event is decided by pipeline timing the + * model cannot see; in lockstep the DUT proves the boundary and the + * driver injects the take there (take_irq/take_debug windows lower + * this hold together with skip_irq_check). + * + * Synchronous conditions keep their architectural timing and ignore + * the hold: the execute-address trigger (evaluated on the matched + * boundary itself), the single-step window close (exactly one + * instruction after dret), ebreak-to-debug and driver-armed entries + * (cause 1/2/4). + * + * This is a LEVEL, unlike exec.skip_irq_check (a one-shot consumed at + * the first check() of a step quantum): a 20 ns engine quantum runs + * several dispatches, and every dispatch after the first ran with the + * one-shot already consumed - the window through which the model used + * to take wire IRQs on its own, racing the DUT's entry boundary. + * Standalone (non co-sim) runs never set it: behaviour unchanged. */ + bool dpi_async_hold = false; + vp::WireSlave haltreq_itf; vp::WireSlave wfi_wake_itf; diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index 4ca5e907..ce6e15b0 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -126,6 +126,17 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) this->declare_csr(&this->minstreth, "minstreth", 0xB82); #endif + /* minstret/minstreth writes go through the default masked store; the + * callbacks (return true) only arm the same-row increment suppression + * consumed by hpm_commit (RTL: the write wins over the increment in + * the writing instruction's own retire cycle). */ + this->minstret.register_callback(std::bind(&Cv32e40pCsr::minstret_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +#if ISS_REG_WIDTH == 32 + this->minstreth.register_callback(std::bind(&Cv32e40pCsr::minstreth_access, this, + std::placeholders::_1, std::placeholders::_2, std::placeholders::_3)); +#endif + /* mcycle/mcycleh: one 64-bit count derived from the clock with a write * offset, frozen into the register pair while mcountinhibit.CY is set. * Registered after the base callback so these have the last word. */ @@ -320,6 +331,7 @@ void Cv32e40pCsr::reset(bool active) this->dcsr = (4 << 28) | 0x3; this->mcycle_offset = 0; + this->minstret_written = false; /* Excluded from the base reset sweep, which walks the CSR map: * mip left it for the mip_view front-end, hwloop_lpend is a plain @@ -486,6 +498,27 @@ bool Cv32e40pCsr::mcycle_access(iss_insn_t *insn, bool is_write, iss_reg_t &valu return false; } +bool Cv32e40pCsr::minstret_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + /* Arm the same-row increment suppression; the store itself is the + * default masked one (return true). */ + if (is_write) + { + this->minstret_written = true; + } + return true; +} + +bool Cv32e40pCsr::minstreth_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) +{ + /* RTL suppresses the increment on a write to EITHER half. */ + if (is_write) + { + this->minstret_written = true; + } + return true; +} + bool Cv32e40pCsr::mcycleh_access(iss_insn_t *insn, bool is_write, iss_reg_t &value) { if (is_write) diff --git a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp index bcb5ccda..2c3b3d7e 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/irq.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/irq.cpp @@ -213,8 +213,11 @@ int Cv32e40pIrq::check() /* Held-high haltreq re-arms (RTL debug_req_i is level-sensitive: the * hart re-halts right after dret while the line stays asserted; the - * wire itself only syncs on level changes). */ - if (this->haltreq_level && !this->req_debug && !this->iss.exec.debug_mode) + * wire itself only syncs on level changes). Under the co-sim hold the + * re-arm waits for the driver's injection window: the level is state, + * the halt boundary is the DUT's to prove (dpi_async_hold contract). */ + if (this->haltreq_level && !this->dpi_async_hold && + !this->req_debug && !this->iss.exec.debug_mode) { this->req_debug = true; /* req_debug_cause keeps its default (3) */ } @@ -244,6 +247,19 @@ int Cv32e40pIrq::check() * when armed by an external debug-entry request. */ if (this->req_debug && !this->iss.exec.debug_mode) { + /* Wire-armed haltreq entry (cause 3) under the co-sim hold: stay + * latched. The haltreq_sync edge arms req_debug at net-delivery + * time, which is row-granular - taking at the next dispatch would + * race the DUT's own halt boundary (the same race as the interrupt + * ladder below). The request is not lost: the driver's take_debug + * window lowers the hold and the entry lands on the DUT-proven + * boundary. Driver-armed causes (1 ebreak / 4 step) only ever run + * inside such a window; the synchronous trigger match (cause 2, + * set above) keeps its architectural boundary and enters here. */ + if (this->dpi_async_hold && this->req_debug_cause == 3 && !trigger_match) + { + return 0; + } /* Informed interrupt+debug collision: the RTL takes the interrupt * first and enters debug on the first handler instruction - dpc is * the (vectored) entry and mstatus/mepc/mcause carry the take; the @@ -330,6 +346,15 @@ int Cv32e40pIrq::check() return 0; } + /* Co-sim hold: pending wired interrupts stay pending. The ladder take + * below picks a boundary out of the model's own stepping cadence, which + * races the DUT's controller timing; the lockstep driver injects the + * take at the DUT-proven entry boundary instead (take_irq window). */ + if (this->dpi_async_hold) + { + return 0; + } + /* M-mode only core: the take needs a wired pending line and the global * enable (mstatus.MIE). */ iss_reg_t pending = this->iss.csr.mie.value & this->iss.csr.mip.value & IRQ_MASK; From f8e71701555f4b858f657d8d1f4ffd1421845883 Mon Sep 17 00:00:00 2001 From: mpaci Date: Mon, 17 Aug 2026 01:54:20 +0200 Subject: [PATCH 26/28] fix(cv32e40p): zero the undeclared S-mode CSR backing; sret raises illegal Undeclaring a CSR removes it from Csr::reset() coverage - reset walks the declared map only - while its raw .value, never initialized by the CsrReg constructor, is still read unguarded by shared trap code: Exception::raise consults medeleg for delegation and redirects through stvec, Core::sret_handle returns sepc. Left as heap garbage, a stray medeleg bit silently delegated synchronous traps to S-mode: mstatus.spp set, mcause/mepc left stale, entry through a garbage stvec. Zero the backing fields once at construction; nothing can write them afterwards, since an undeclared CSR raises illegal on any ISA access, like the RTL. sret itself now raises illegal instruction instead of running the generic handler: CV32E40P has no S-mode and the RTL decoder rejects the opcode. The generic path would jump through sepc - never architecturally written on this core - and demote the privilege mode via mstatus.spp. Same guard shape as dret_exec outside debug mode. --- cpu/iss_v2/include/cores/cv32e40p/priv.hpp | 8 ++++++-- cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 24 ++++++++++++++++++++++ 2 files changed, 30 insertions(+), 2 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp index 615a6faa..1e736186 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/priv.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/priv.hpp @@ -230,8 +230,12 @@ static inline iss_reg_t dret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) static inline iss_reg_t sret_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) { - iss->timing.stall_insn_dependency_account(5); - return iss->core.sret_handle(); + /* No S-mode on CV32E40P: the RTL decodes sret as illegal. The generic + * handler would jump through sepc - never architecturally written on + * this core - and demote the privilege mode via mstatus.spp. Same + * guard shape as dret_exec above. */ + iss->exception.raise(pc, ISS_EXCEPT_ILLEGAL); + return pc; } static inline iss_reg_t sfence_vma_exec(Iss *iss, iss_insn_t *insn, iss_reg_t pc) diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index ce6e15b0..b1b9709b 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -103,6 +103,30 @@ Cv32e40pCsr::Cv32e40pCsr(Iss &iss) this->undeclare_csr(addr); } + /* Undeclaring removes a register from Csr::reset() coverage (reset walks + * the declared map only) while its raw .value - never initialized by the + * CsrReg constructor - is still read unguarded by shared trap code: + * Exception::raise consults medeleg for delegation and redirects through + * stvec; Core::sret_handle returns sepc. Left as heap garbage, a stray + * medeleg bit silently delegated sync traps to S-mode: mstatus.spp set + * (the bit-8 delta of the C3 trap-snapshot lanes), mcause/mepc stale, + * entry at garbage stvec (the 0x20202020 runaways). Zero them once here; + * nothing can write them afterwards - undeclared means any ISA access + * raises illegal, like the RTL. satp is left alone: guarded by + * CONFIG_GVSOC_ISS_MMU on every read path and unreachable on this core. */ + this->medeleg.value = 0; + this->mideleg.value = 0; + this->sstatus.value = 0; + this->sie.value = 0; + this->stvec.value = 0; + this->scounteren.value = 0; + this->sscratch.value = 0; + this->sepc.value = 0; + this->scause.value = 0; + this->stval.value = 0; + this->sip.value = 0; + this->mcounteren.value = 0; + this->raise_on_unsupported_csr = true; /* Machine information registers: read-only, writes raise illegal. From 8a5b299275f55d1f9a70e191fce386544e1dabc7 Mon Sep 17 00:00:00 2001 From: mpaci Date: Mon, 17 Aug 2026 01:54:56 +0200 Subject: [PATCH 27/28] fix(cv32e40p): drop writes to x0 - the XPULP post-increment corrupted it x0 is hardwired to zero (cv32e40p_register_file_ff.sv "R0 is nil"; unpriv spec). The decoder redirects rd==x0 writes to ISS_DUMMY_REG, but the XPULP post-increment addressing modes write the base register back through in_regs (IN_REG_SET, corev.hpp), which is never remapped: cv.lbu x30,(x0),x14 computed the load at address 0 correctly and then executed x0 += x14 in the model, while the RTL register file discards the write. The stray write is invisible at the writing instruction (the architecturally written register is x30, and it is correct); everything downstream that reads x0 then diverges. Writes to x0 are architectural no-ops: drop them up front in set_reg, so they cannot consume the wb_suppress one-shot either. The v1 ISS has the same defect (regfile_implem.hpp:55, and pulp_v2.hpp writes in_regs back at 32 sites without even the rd==rs1 guard): it affects ri5cy standalone runs and is a candidate upstream report. --- cpu/iss_v2/include/cores/cv32e40p/regfile.hpp | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp b/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp index 4f4b9c0e..09c86d6f 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/regfile.hpp @@ -33,6 +33,18 @@ class Cv32e40pRegfile : public Regfile /* Shadows the base setters (static dispatch via CONFIG_GVSOC_ISS_REGFILE). */ inline void set_reg(int reg, uint64_t value) { + /* x0 is hardwired to zero (RTL: cv32e40p_register_file_ff.sv "R0 is + * nil"; unpriv spec). The decoder redirects rd==x0 writes to + * ISS_DUMMY_REG, but the XPULP post-increment addressing modes write + * the base register back through in_regs (IN_REG_SET, corev.hpp), + * which is never remapped: with rs1==x0 the increment lands in the + * real x0 slot and corrupts it (class C11, fv_ms1_20260816). Writes + * to x0 are architectural no-ops: drop them up front so they can + * never consume the wb_suppress one-shot either. */ + if (reg == 0) + { + return; + } if (this->wb_suppress) { this->wb_suppress = false; From f5764c76f8ba12c7c986b44d3f81665f2034f08b Mon Sep 17 00:00:00 2001 From: mpaci Date: Tue, 18 Aug 2026 14:10:55 +0200 Subject: [PATCH 28/28] fix(cv32e40p): gate hpm increments on the pre-write mcountinhibit for the writing instruction The RTL evaluates the counter increment gates on mcountinhibit_q in the cycle the CSR write commits (cv32e40p_cs_registers.sv:1428), so the instruction that writes mcountinhibit still counts under the OLD gates and the write takes effect from the next instruction on. Latch the pre-write value when the write executes and let hpm_commit consume it at that same retire; both flags clear unconditionally, they belong to that retire only. --- cpu/iss_v2/include/cores/cv32e40p/csr.hpp | 22 ++++++++++++++++++---- cpu/iss_v2/src/cores/cv32e40p/csr.cpp | 5 +++++ 2 files changed, 23 insertions(+), 4 deletions(-) diff --git a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp index d1b98038..828cd6d2 100644 --- a/cpu/iss_v2/include/cores/cv32e40p/csr.hpp +++ b/cpu/iss_v2/include/cores/cv32e40p/csr.hpp @@ -223,6 +223,15 @@ class Cv32e40pCsr : public Csr * (cv32e40p_cs_registers.sv, !write_lower && !write_upper gate), so * the csrw itself must not count on top of the written value. */ bool minstret_written = false; + + /* Armed by a CSR write to mcountinhibit, consumed (and cleared) by + * hpm_commit at that same instruction's retire: the RTL evaluates the + * increment gates on mcountinhibit_q in the cycle the write commits + * (cv32e40p_cs_registers.sv:1428), so the writing instruction is still + * gated by the OLD value and the write takes effect from the next + * instruction on. */ + bool mcountinhibit_stale = false; + iss_reg_t mcountinhibit_old = 0; }; inline bool Cv32e40pCsr::fp_access_illegal() @@ -249,13 +258,18 @@ inline bool Cv32e40pCsr::hpm_counting() inline void Cv32e40pCsr::hpm_commit(uint32_t events, bool count_instr) { + /* An instruction writing mcountinhibit is gated by the pre-write + * value; both flags clear unconditionally: they belong to this + * retire only. */ + iss_reg_t inhibit = this->mcountinhibit_stale ? this->mcountinhibit_old + : this->mcountinhibit.value; + this->mcountinhibit_stale = false; /* minstret: retired instructions, gated on mcountinhibit.IR (bit 2), * on the RTL event line (count_instr, false for EBREAK) and on the - * same-row write suppression. The flag clears unconditionally: it - * belongs to this retire only. */ + * same-row write suppression. */ bool wrote_counter = this->minstret_written; this->minstret_written = false; - if (count_instr && !wrote_counter && !(this->mcountinhibit.value & 0x4)) + if (count_instr && !wrote_counter && !(inhibit & 0x4)) { if (++this->minstret.value == 0) { @@ -269,7 +283,7 @@ inline void Cv32e40pCsr::hpm_commit(uint32_t events, bool count_instr) for (int i = 0; i < CONFIG_GVSOC_ISS_CV32E40P_NUM_MHPMCOUNTERS; i++) { if ((this->mhpmevent[i].value & events) - && !(this->mcountinhibit.value & (1u << (3 + i)))) + && !(inhibit & (1u << (3 + i)))) { if (++this->mhpmcounter[i].value == 0) { diff --git a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp index b1b9709b..60b85893 100644 --- a/cpu/iss_v2/src/cores/cv32e40p/csr.cpp +++ b/cpu/iss_v2/src/cores/cv32e40p/csr.cpp @@ -356,6 +356,7 @@ void Cv32e40pCsr::reset(bool active) this->mcycle_offset = 0; this->minstret_written = false; + this->mcountinhibit_stale = false; /* Excluded from the base reset sweep, which walks the CSR map: * mip left it for the mip_view front-end, hwloop_lpend is a plain @@ -621,6 +622,10 @@ bool Cv32e40pCsr::mcountinhibit_access(iss_insn_t *insn, bool is_write, iss_reg_ /* Event lines fire only from the full handlers (same scheme as the * Ri5ky PCMR write): switch when software touches the inhibit CSR. */ this->iss.exec.switch_to_full_mode(); + /* The writing instruction itself still counts under the OLD gates + * (consumed by hpm_commit at this same retire). */ + this->mcountinhibit_old = this->mcountinhibit.value; + this->mcountinhibit_stale = true; bool old_cy = this->mcountinhibit.value & 0x1; bool new_cy = value & 0x1; if (old_cy != new_cy)